{"data":[{"id":"deepseek-v4-flash","name":"DeepSeek: DeepSeek V4 Flash 0423","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.6247000000000002e-07,"completion":3.2494000000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.0800000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.17036214272000003,"max_completion_cost":0.12777160704,"max_cost":0.23424794624000003},"sats_pricing":{"prompt":0.00024989886840279756,"completion":0.0004997977368055951,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.737419306214172e-05,"input_cache_write":0.0,"max_prompt_cost":262.03795583433185,"max_completion_cost":196.52846687574888,"max_cost":360.3021892722063},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v4-flash-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-pro","name":"DeepSeek: DeepSeek V4 Pro","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":5.048175000000001e-07,"completion":1.0096350000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.9875e-09,"input_cache_write":0.0,"max_prompt_cost":0.5293395148800001,"max_completion_cost":0.3876998400000001,"max_cost":0.7231894348800001},"sats_pricing":{"prompt":0.0007764714839658352,"completion":0.0015529429679316704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.1332660660808465e-06,"input_cache_write":0.0,"max_prompt_cost":814.1893627709596,"max_completion_cost":596.3300996857615,"max_cost":1112.3544126138404},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v4-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5","name":"Claude Opus 5","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":2.9012500000000004e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.802500000000001,"max_completion_cost":3.7136000000000005,"max_cost":8.773380000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.044624797929070995,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":8924.959585814198,"max_completion_cost":5711.974134921087,"max_cost":13494.538893751069},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol","name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":6.092625000000001,"max_completion_cost":4.45632,"max_cost":9.806225000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":9371.207565104909,"max_completion_cost":6854.368961905303,"max_cost":15083.181700025994},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.5","name":"SpaceXAI: Grok 4.5","created":1783523154,"description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":3.3e-07,"input_cache_write":0.0,"max_prompt_cost":1.1604999999999999,"max_completion_cost":3.4815000000000005,"max_cost":3.4815000000000005},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0005075806399515183,"input_cache_write":0.0,"max_prompt_cost":1784.9919171628392,"max_completion_cost":5354.975751488519,"max_cost":5354.975751488519},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.5-20260708","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5","name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":2.7500000000000004e-06,"max_prompt_cost":2.3209999999999997,"max_completion_cost":1.48544,"max_cost":3.509352},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.004229838666262653,"max_prompt_cost":3569.9838343256783,"max_completion_cost":2284.789653968435,"max_cost":5397.815557500426},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.2494000000000005e-07,"completion":1.0212400000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.204200000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.33273856000000007,"max_completion_cost":0.13071872,"max_cost":0.42186496000000007},"sats_pricing":{"prompt":0.0004997977368055951,"completion":0.001570792887103299,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.466577352982344e-05,"input_cache_write":0.0,"max_prompt_cost":511.7928824889294,"max_completion_cost":201.06148954922224,"max_cost":648.8802617270355},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5","name":"Anthropic: Claude Fable 5","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":5.802500000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-06,"input_cache_write":1.3750000000000002e-05,"max_prompt_cost":11.605000000000002,"max_completion_cost":7.427200000000001,"max_cost":17.546760000000003},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.08924959585814199,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0016919354665050612,"input_cache_write":0.021149193331313265,"max_prompt_cost":17849.919171628397,"max_completion_cost":11423.948269842174,"max_cost":26989.077787502138},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro","name":"OpenAI: GPT-5.5 Pro","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.4815e-05,"completion":0.00020889000000000001,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-06,"input_cache_write":0.0,"max_prompt_cost":36.555749999999996,"max_completion_cost":26.737920000000003,"max_cost":58.83735},"sats_pricing":{"prompt":0.05354975751488518,"completion":0.3212985450893111,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0050758063995151835,"input_cache_write":0.0,"max_prompt_cost":56227.24539062943,"max_completion_cost":41126.21377143182,"max_cost":90499.09020015596},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":1.4506250000000002e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.09284,"max_completion_cost":0.18568,"max_cost":0.2488112},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.0022312398964535497,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":3.3838709330101225e-05,"input_cache_write":0.0,"max_prompt_cost":142.79935337302717,"max_completion_cost":285.59870674605435,"max_cost":382.7022670397128},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.703750000000001e-07,"completion":5.22225e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.34815000000000007,"max_completion_cost":0.6684479999999999,"max_cost":0.9051899999999999},"sats_pricing":{"prompt":0.0013387439378721297,"completion":0.008032463627232778,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00012689515998787958,"input_cache_write":0.0,"max_prompt_cost":535.497575148852,"max_completion_cost":1028.1553442857953,"max_cost":1392.2936953870146},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-chat","name":"OpenAI: GPT-5.3 Chat","created":1772564061,"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.030875e-06,"completion":1.6247e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.925e-07,"input_cache_write":0.0,"max_prompt_cost":0.25995199999999996,"max_completion_cost":0.266190848,"max_cost":0.49286899199999995},"sats_pricing":{"prompt":0.003123735855034969,"completion":0.02498988684027975,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0002960887066383857,"input_cache_write":0.0,"max_prompt_cost":399.838189444476,"max_completion_cost":409.43430599114345,"max_cost":758.0932071867264},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.3-chat-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.3-codex","name":"OpenAI: GPT-5.3-Codex","created":1771959164,"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.030875e-06,"completion":1.6247e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.925e-07,"input_cache_write":0.0,"max_prompt_cost":0.8123499999999999,"max_completion_cost":2.0796159999999997,"max_cost":2.632014},"sats_pricing":{"prompt":0.003123735855034969,"completion":0.02498988684027975,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0002960887066383857,"input_cache_write":0.0,"max_prompt_cost":1249.4943420139875,"max_completion_cost":3198.705515555808,"max_cost":4048.3616681253197},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.3-codex-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview","name":"Google: Gemini 3 Flash Preview","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.85e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":5.5e-07,"web_search":0.015400000000000002,"internal_reasoning":3.3e-06,"input_cache_read":5.5e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.40370176,"max_completion_cost":0.15138816000000002,"max_cost":0.5298585600000001},"sats_pricing":{"prompt":0.0005921774132767714,"completion":0.003553064479660629,"request":0.001,"image":0.0008459677332525306,"web_search":23.687096531070857,"internal_reasoning":0.0050758063995151835,"input_cache_read":8.459677332525305e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":620.9430233041038,"max_completion_cost":232.85363373903897,"max_cost":814.9877180866364},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":5.802500000000001e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":0.2321,"max_completion_cost":0.37136,"max_cost":0.529188},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.008924959585814199,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0021149193331313266,"max_prompt_cost":356.9983834325679,"max_completion_cost":571.1974134921087,"max_cost":813.9563142262548},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"ling-3.0-tiny","name":"Ling 3.0 Tiny","created":1786034890,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.09126543360000001,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":140.37747633982065,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"muse-spark-1.2","name":"Meta: Muse Spark 1.2","created":1785959287,"description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":4.932125000000001e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":1.5210905600000002,"max_completion_cost":5.171707904000001,"max_cost":5.171707904000001},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.007586215647942068,"request":0.001,"image":0.0,"web_search":4.229838666262653,"internal_reasoning":0.0,"input_cache_read":0.00025379031997575915,"input_cache_write":0.0,"max_prompt_cost":2339.6246056636774,"max_completion_cost":7954.723659256502,"max_cost":7954.723659256502},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta/muse-spark-1.2-20260805","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.8-max","name":"Qwen: Qwen3.8 Max","created":1785731612,"description":"Qwen3.8 Max is the flagship model in Alibaba's Qwen3.8 series, the general-availability successor to the Qwen3.8 Max Preview. It is a multimodal reasoning model intended for complex reasoning, visual understanding,...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":2.7500000000000004e-06,"max_prompt_cost":2.3209999999999997,"max_completion_cost":0.9126543360000001,"max_cost":2.929436224},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004229838666262653,"input_cache_write":0.004229838666262653,"max_prompt_cost":3569.9838343256783,"max_completion_cost":1403.7747633982062,"max_cost":4505.833676591149},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.8-max-20260803","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","created":1785606009,"description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.0444500000000002e-07,"completion":2.0889000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.9712e-08,"input_cache_write":0.0,"max_prompt_cost":0.10951852032000002,"max_completion_cost":0.027379630080000005,"max_cost":0.12320833536000003},"sats_pricing":{"prompt":0.00016064927254465559,"completion":0.00032129854508931117,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.0319483559770695e-05,"input_cache_write":0.0,"max_prompt_cost":168.45297160778478,"max_completion_cost":42.113242901946194,"max_cost":189.50959305875787},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~deepseek/deepseek-v4-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v4-flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","created":1785478908,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":1.0444500000000002e-07,"completion":2.0889000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.9800000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.10951852032000002,"max_completion_cost":0.08021376000000001,"max_cost":0.14962540032000002},"sats_pricing":{"prompt":0.00016064927254465559,"completion":0.00032129854508931117,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.04548383970911e-05,"input_cache_write":0.0,"max_prompt_cost":168.45297160778478,"max_completion_cost":123.37864131429548,"max_cost":230.1422922649325},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":384000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v4-flash-20260731","alias_ids":null,"forwarded_model_id":null},{"id":"inkling-small","name":"Thinking Machines: Inkling Small","created":1785443117,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.22225e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.2737963008,"max_completion_cost":0.36506173440000006,"max_cost":0.5019598848000001},"sats_pricing":{"prompt":0.0008032463627232777,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":421.1324290194618,"max_completion_cost":561.5099053592826,"max_cost":772.0761198690135},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thinkingmachines/inkling-small-20260730","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-flash","name":"Qwen: Qwen3.7 Flash","created":1785190561,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.4815e-08,"completion":1.5086500000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-09,"input_cache_write":4.1800000000000004e-08,"max_prompt_cost":0.034815,"max_completion_cost":0.009887088640000001,"max_cost":0.042420452799999994},"sats_pricing":{"prompt":5.354975751488518e-05,"completion":0.00023204894923116914,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0151612799030368e-05,"input_cache_write":6.429354772719233e-05,"max_prompt_cost":53.54975751488518,"max_completion_cost":15.207559936813901,"max_cost":65.24788054320356},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.7-flash-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5-fast","name":"Claude Opus 5 (Fast)","created":1784912546,"description":"Fast-mode variant of [Opus 5](/anthropic/claude-opus-5) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 5.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":5.802500000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-06,"input_cache_write":1.3750000000000002e-05,"max_prompt_cost":11.605000000000002,"max_completion_cost":7.427200000000001,"max_cost":17.546760000000003},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.08924959585814199,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0016919354665050612,"input_cache_write":0.021149193331313265,"max_prompt_cost":17849.919171628397,"max_completion_cost":11423.948269842174,"max_cost":26989.077787502138},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-opus-5-fast-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"ling-3.0-flash","name":"Ling-3.0-flash","created":1784818580,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.4370500000000003e-08,"completion":7.31115e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.620000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.006388580352000001,"max_completion_cost":0.002395717632,"max_cost":0.00798572544},"sats_pricing":{"prompt":3.7484830260419635e-05,"completion":0.00011245449078125889,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.106128959321258e-06,"input_cache_write":0.0,"max_prompt_cost":9.826423343787445,"max_completion_cost":3.684908753920291,"max_cost":12.283029179734305},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ling-3.0-flash-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-s-2.1","name":"Poolside: Laguna S 2.1","created":1784652683,"description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.0444500000000002e-07,"completion":2.0889000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.900000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.10951852032000002,"max_completion_cost":0.027379630080000005,"max_cost":0.12320833536000003},"sats_pricing":{"prompt":0.00016064927254465559,"completion":0.00032129854508931117,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.522741919854555e-05,"input_cache_write":0.0,"max_prompt_cost":168.45297160778478,"max_completion_cost":42.113242901946194,"max_cost":189.50959305875787},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"poolside/laguna-s-2.1-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash","name":"Google: Gemini 3.6 Flash","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":5.775e-06,"request":0.0,"image":1.65e-06,"web_search":0.015400000000000002,"internal_reasoning":8.25e-06,"input_cache_read":1.65e-07,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":1.2111052800000002,"max_completion_cost":0.3784704,"max_cost":1.5138816},"sats_pricing":{"prompt":0.0017765322398303144,"completion":0.00888266119915157,"request":0.001,"image":0.0025379031997575918,"web_search":23.687096531070857,"internal_reasoning":0.012689515998787959,"input_cache_read":0.00025379031997575915,"input_cache_write":0.00014099462220875506,"max_prompt_cost":1862.8290699123118,"max_completion_cost":582.1340843475973,"max_cost":2328.5363373903892},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite","name":"Google: Gemini 3.5 Flash Lite","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.3100000000000002e-07,"completion":1.925e-06,"request":0.0,"image":3.3e-07,"web_search":0.015400000000000002,"internal_reasoning":2.7500000000000004e-06,"input_cache_read":3.3e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.24222105600000002,"max_completion_cost":0.1261568,"max_cost":0.35323904000000006},"sats_pricing":{"prompt":0.00035530644796606285,"completion":0.0029608870663838573,"request":0.001,"image":0.0005075806399515183,"web_search":23.687096531070857,"internal_reasoning":0.004229838666262653,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":372.5658139824623,"max_completion_cost":194.04469478253247,"max_cost":543.3251453910909},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"longcat-2.0","name":"Meituan: LongCat 2.0","created":1784554658,"description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","context_length":1048756,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-09,"input_cache_write":0.0,"max_prompt_cost":0.3651244014000001,"max_completion_cost":0.36506173440000006,"max_cost":0.6389207022000001},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0151612799030368e-05,"input_cache_write":0.0,"max_prompt_cost":561.6062949228094,"max_completion_cost":561.5099053592826,"max_cost":982.7387239422712},"per_request_limits":null,"top_provider":{"context_length":1048756,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meituan/longcat-2.0-20260720","alias_ids":null,"forwarded_model_id":null},{"id":"inkling","name":"Thinking Machines: Inkling","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":1048576,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1024750000000002e-06,"completion":4.700025e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.7600000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":0.5780144128000001,"max_completion_cost":1.2320833536,"max_cost":1.52109056},"sats_pricing":{"prompt":0.0016957423213046978,"completion":0.0072292172645095,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002707096746408098,"input_cache_write":0.0,"max_prompt_cost":889.0573501521974,"max_completion_cost":1895.0959305875783,"max_cost":2339.624605663677},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k3","name":"MoonshotAI: Kimi K3","created":1784215858,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-07,"input_cache_write":0.0,"max_prompt_cost":3.6506173440000005,"max_completion_cost":18.25308672,"max_cost":18.25308672},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005075806399515183,"input_cache_write":0.0,"max_prompt_cost":5615.099053592825,"max_completion_cost":28075.49526796412,"max_cost":28075.49526796412},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k3-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"muse-spark-1.1","name":"Meta: Muse Spark 1.1","created":1784215741,"description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":4.932125000000001e-06,"request":0.0,"image":0.0,"web_search":0.0027500000000000003,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":1.5210905600000002,"max_completion_cost":5.171707904000001,"max_cost":5.171707904000001},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.007586215647942068,"request":0.001,"image":0.0,"web_search":4.229838666262653,"internal_reasoning":0.0,"input_cache_read":0.00025379031997575915,"input_cache_write":0.0,"max_prompt_cost":2339.6246056636774,"max_completion_cost":7954.723659256502,"max_cost":7954.723659256502},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta/muse-spark-1.1-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-air-v2.5","name":"Kwaipilot: KAT-Coder-Air V2.5","created":1783714590,"description":"KAT-Coder-Air V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.044563200000000004,"max_completion_cost":0.05570400000000001,"max_cost":0.0863412},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":68.54368961905304,"max_completion_cost":85.6796120238163,"max_cost":132.80339863691526},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"kwaipilot/kat-coder-air-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","created":1783714589,"description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.587700000000001e-07,"completion":3.4350800000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":0.21984512,"max_completion_cost":0.2748064,"max_cost":0.42594992},"sats_pricing":{"prompt":0.0013208940187005012,"completion":0.005283576074802005,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00025379031997575915,"input_cache_write":0.0,"max_prompt_cost":338.1488687873283,"max_completion_cost":422.6860859841604,"max_cost":655.1634332754486},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"kwaipilot/kat-coder-pro-v2.5-20260710","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro","name":"OpenAI: GPT-5.6 Luna Pro","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":1.375e-07,"max_prompt_cost":0.12185250000000002,"max_completion_cost":0.08912640000000001,"max_cost":0.19612450000000003},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":1.6919354665050612e-05,"input_cache_write":0.00021149193331313266,"max_prompt_cost":187.42415130209815,"max_completion_cost":137.08737923810608,"max_cost":301.6636340005199},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna","name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":1.375e-07,"max_prompt_cost":0.12185250000000002,"max_completion_cost":0.08912640000000001,"max_cost":0.19612450000000003},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":1.6919354665050612e-05,"input_cache_write":0.00021149193331313266,"max_prompt_cost":187.42415130209815,"max_completion_cost":137.08737923810608,"max_cost":301.6636340005199},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro","name":"OpenAI: GPT-5.6 Terra Pro","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":1.2185249999999999,"max_completion_cost":0.8912640000000002,"max_cost":1.961245},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0021149193331313266,"max_prompt_cost":1874.241513020981,"max_completion_cost":1370.8737923810609,"max_cost":3016.6363400051987},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra","name":"OpenAI: GPT-5.6 Terra","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":1.2185249999999999,"max_completion_cost":0.8912640000000002,"max_cost":1.961245},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0021149193331313266,"max_prompt_cost":1874.241513020981,"max_completion_cost":1370.8737923810609,"max_cost":3016.6363400051987},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro","name":"OpenAI: GPT-5.6 Sol Pro","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":6.092625000000001,"max_completion_cost":4.45632,"max_cost":9.806225000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":9371.207565104909,"max_completion_cost":6854.368961905303,"max_cost":15083.181700025994},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"grok-latest","name":"xAI: Grok Latest","created":1783519360,"description":"This model always redirects to the latest Grok model from xAI.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":3.3e-07,"input_cache_write":0.0,"max_prompt_cost":1.1604999999999999,"max_completion_cost":3.4815000000000005,"max_cost":3.4815000000000005},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0005075806399515183,"input_cache_write":0.0,"max_prompt_cost":1784.9919171628392,"max_completion_cost":5354.975751488519,"max_cost":5354.975751488519},"per_request_limits":null,"top_provider":{"context_length":500000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~x-ai/grok-latest","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","created":1783443096,"description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.1235e-07,"completion":1.6247e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.9800000000000003e-07,"input_cache_write":0.0,"max_prompt_cost":0.1064763392,"max_completion_cost":0.0532381696,"max_cost":0.13309542400000002},"sats_pricing":{"prompt":0.0012494943420139877,"completion":0.0024989886840279755,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00030454838397091103,"input_cache_write":0.0,"max_prompt_cost":163.7737223964574,"max_completion_cost":81.8868611982287,"max_cost":204.71715299557175},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-3.0-mini-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"aion-3.0","name":"AionLabs: Aion-3.0","created":1783443095,"description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-07,"input_cache_write":0.0,"max_prompt_cost":0.45632716800000006,"max_completion_cost":0.22816358400000003,"max_cost":0.5704089600000001},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0012689515998787959,"input_cache_write":0.0,"max_prompt_cost":701.8873816991031,"max_completion_cost":350.94369084955156,"max_cost":877.3592271238789},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-3.0-20260707","alias_ids":null,"forwarded_model_id":null},{"id":"hy3","name":"Tencent: Hy3","created":1783344048,"description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.53186e-07,"completion":6.12744e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.63e-08,"input_cache_write":0.0,"max_prompt_cost":0.040156790784,"max_completion_cost":0.078431232,"max_cost":0.098980214784},"sats_pricing":{"prompt":0.00023561893306549482,"completion":0.0009424757322619793,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.583387039466702e-05,"input_cache_write":0.0,"max_prompt_cost":61.76608958952107,"max_completion_cost":120.63689372953334,"max_cost":152.24375988667109},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"tencent/hy3-20260706","alias_ids":null,"forwarded_model_id":null},{"id":"laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","created":1783002429,"description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963e-08,"completion":1.3926e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.01825308672,"max_completion_cost":0.00456327168,"max_cost":0.020534722559999996},"sats_pricing":{"prompt":0.00010709951502977036,"completion":0.00021419903005954073,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":28.075495267964122,"max_completion_cost":7.0188738169910305,"max_cost":31.58493217645963},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"poolside/laguna-xs-2.1-20260625","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-image","name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","created":1782837225,"description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.925e-07,"completion":1.1550000000000002e-06,"request":0.0,"image":0.0,"web_search":0.015400000000000002,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.01261568,"max_completion_cost":0.07569408000000001,"max_cost":0.07569408000000001},"sats_pricing":{"prompt":0.0002960887066383857,"completion":0.0017765322398303144,"request":0.001,"image":0.0,"web_search":23.687096531070857,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.404469478253244,"max_completion_cost":116.42681686951948,"max_cost":116.42681686951948},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-lite-image-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-mini","name":"Nex AGI: Nex-N2-Mini","created":1782312964,"description":"Nex-N2-Mini is an open-source agentic mixture-of-experts model from Nex AGI, the smaller sibling in the Nex-N2 series. It accepts text and image input and is built for coding, tool use,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.9012500000000003e-08,"completion":1.1605000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.7500000000000002e-09,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.030421811200000003,"max_cost":0.030421811200000003},"sats_pricing":{"prompt":4.462479792907099e-05,"completion":0.00017849919171628395,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.229838666262653e-06,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":46.79249211327354,"max_cost":46.79249211327354},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nex-agi/nex-n2-mini","alias_ids":null,"forwarded_model_id":null},{"id":"fugu-ultra","name":"Sakana: Fugu Ultra","created":1782276303,"description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":0.0,"max_prompt_cost":5.802500000000001,"max_completion_cost":4.45632,"max_cost":9.516100000000002},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.0,"max_prompt_cost":8924.959585814198,"max_completion_cost":6854.368961905303,"max_cost":14636.933720735286},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sakana/fugu-ultra-20260615","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","created":1781754065,"description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.85e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.015400000000000002,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05046272,"max_completion_cost":0.07569408000000001,"max_cost":0.11354112000000001},"sats_pricing":{"prompt":0.0005921774132767714,"completion":0.003553064479660629,"request":0.001,"image":0.0,"web_search":23.687096531070857,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":77.61787791301298,"max_completion_cost":116.42681686951948,"max_cost":174.6402253042792},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image","name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","created":1781754054,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.54e-06,"completion":9.240000000000001e-06,"request":0.0,"image":2.2e-06,"web_search":0.015400000000000002,"internal_reasoning":1.32e-05,"input_cache_read":2.2e-07,"input_cache_write":4.125e-07,"max_prompt_cost":0.10092544,"max_completion_cost":0.30277632000000004,"max_cost":0.35323904000000006},"sats_pricing":{"prompt":0.0023687096531070854,"completion":0.014212257918642515,"request":0.001,"image":0.0033838709330101225,"web_search":23.687096531070857,"internal_reasoning":0.020303225598060734,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0006344757999393979,"max_prompt_cost":155.23575582602595,"max_completion_cost":465.70726747807794,"max_cost":543.3251453910909},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3-pro-image-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"north-mini-code","name":"North Mini Code","created":1781723748,"description":"PPQ.AI model","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08912640000000001,"max_completion_cost":0.29708799999999996,"max_cost":0.29708799999999996},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":137.08737923810608,"max_completion_cost":456.95793079368684,"max_cost":456.95793079368684},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code","name":"MoonshotAI: Kimi K2.7 Code","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.1235e-07,"completion":4.06175e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":0.2129526784,"max_completion_cost":1.064763392,"max_cost":1.064763392},"sats_pricing":{"prompt":0.0012494943420139877,"completion":0.006247471710069938,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00025379031997575915,"input_cache_write":0.0,"max_prompt_cost":327.5474447929148,"max_completion_cost":1637.7372239645738,"max_cost":1637.7372239645738},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-latest","name":"Anthropic: Claude Fable Latest","created":1781029944,"description":"This model always redirects to the latest model in the Claude Fable family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":5.802500000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-06,"input_cache_write":1.3750000000000002e-05,"max_prompt_cost":11.605000000000002,"max_completion_cost":7.427200000000001,"max_cost":17.546760000000003},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.08924959585814199,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0016919354665050612,"input_cache_write":0.021149193331313265,"max_prompt_cost":17849.919171628397,"max_completion_cost":11423.948269842174,"max_cost":26989.077787502138},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-fable-latest","alias_ids":null,"forwarded_model_id":null},{"id":"nex-n2-pro","name":"Nex AGI: Nex-N2-Pro","created":1780937140,"description":"Nex-N2-Pro is an agentic mixture-of-experts model from Nex AGI, with 17B active parameters out of 397B total. Built on the Qwen3.5 architecture, it accepts text and image input and produces...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.076054528,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.0,"max_prompt_cost":116.98123028318383,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nex-agi/nex-n2-pro","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","created":1780581864,"description":"PPQ.AI model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.044563200000000004,"max_completion_cost":0.14854399999999998,"max_cost":0.14854399999999998},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":68.54368961905304,"max_completion_cost":228.47896539684342,"max_cost":228.47896539684342},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b","name":"NVIDIA: Nemotron 3 Ultra","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":4.1778e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.35670613440000004,"max_completion_cost":2.1402368064,"max_cost":2.1402368064},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.006425970901786222,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":548.6579635557101,"max_completion_cost":3291.94778133426,"max_cost":3291.94778133426},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-plus","name":"Qwen: Qwen3.7 Plus","created":1780491783,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.7136e-07,"completion":1.48544e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.040000000000001e-08,"input_cache_write":4.4e-07,"max_prompt_cost":0.37136,"max_completion_cost":0.19469959168,"max_cost":0.51738469376},"sats_pricing":{"prompt":0.0005711974134921087,"completion":0.0022847896539684347,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010828386985632393,"input_cache_write":0.0006767741866020244,"max_prompt_cost":571.1974134921087,"max_completion_cost":299.4719495249507,"max_cost":795.8013756358215},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.7-plus-20260602","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3","name":"MiniMax: MiniMax M3","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-08,"input_cache_write":0.0,"max_prompt_cost":0.18253086720000003,"max_completion_cost":0.7130112000000001,"max_cost":0.7172892672000001},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010151612799030365,"input_cache_write":0.0,"max_prompt_cost":280.7549526796413,"max_completion_cost":1096.6990339048486,"max_cost":1103.2792281082777},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":512000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.7-flash","name":"StepFun: Step 3.7 Flash","created":1779985069,"description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":1.334575e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4000000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.05941760000000001,"max_completion_cost":0.3416512,"max_cost":0.3416512},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.0020527407047372655,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.767741866020245e-05,"input_cache_write":0.0,"max_prompt_cost":91.3915861587374,"max_completion_cost":525.50162041274,"max_cost":525.50162041274},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":256000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"stepfun/step-3.7-flash-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8-fast","name":"Anthropic: Claude Opus 4.8 (Fast)","created":1779913703,"description":"Fast-mode variant of [Opus 4.8](/anthropic/claude-opus-4.8) - identical capabilities with higher output speed at 2x pricing relative to regular Opus 4.8.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":5.802500000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-06,"input_cache_write":1.3750000000000002e-05,"max_prompt_cost":11.605000000000002,"max_completion_cost":7.427200000000001,"max_cost":17.546760000000003},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.08924959585814199,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0016919354665050612,"input_cache_write":0.021149193331313265,"max_prompt_cost":17849.919171628397,"max_completion_cost":11423.948269842174,"max_cost":26989.077787502138},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.8-opus-fast-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8","name":"Anthropic: Claude Opus 4.8","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":2.9012500000000004e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.802500000000001,"max_completion_cost":3.7136000000000005,"max_cost":8.773380000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.044624797929070995,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":8924.959585814198,"max_completion_cost":5711.974134921087,"max_cost":13494.538893751069},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.7-max","name":"Qwen: Qwen3.7 Max","created":1779376861,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.7117375000000001e-06,"completion":5.1352125000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.2450000000000003e-07,"input_cache_write":2.028125e-06,"max_prompt_cost":1.7117375000000001,"max_completion_cost":0.6730825728000001,"max_cost":2.1604592152000004},"sats_pricing":{"prompt":0.0026328630778151884,"completion":0.007898589233445564,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.000499120962618993,"input_cache_write":0.0031195060163687065,"max_prompt_cost":2632.8630778151883,"max_completion_cost":1035.283888006177,"max_cost":3323.0523364859732},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.7-max-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"grok-build-0.1","name":"SpaceXAI: Grok Build 0.1","created":1779298123,"description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.29708799999999996,"max_completion_cost":0.5941759999999999,"max_cost":0.5941759999999999},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":456.95793079368684,"max_completion_cost":913.9158615873737,"max_cost":913.9158615873737},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-build-0.1-20260520","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash","name":"Google: Gemini 3.5 Flash","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":6.9300000000000006e-06,"request":0.0,"image":1.65e-06,"web_search":0.015400000000000002,"internal_reasoning":9.900000000000002e-06,"input_cache_read":1.65e-07,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":1.2111052800000002,"max_completion_cost":0.45416448000000004,"max_cost":1.58957568},"sats_pricing":{"prompt":0.0017765322398303144,"completion":0.010659193438981885,"request":0.001,"image":0.0025379031997575918,"web_search":23.687096531070857,"internal_reasoning":0.015227419198545552,"input_cache_read":0.00025379031997575915,"input_cache_write":0.00014099462220875506,"max_prompt_cost":1862.8290699123118,"max_completion_cost":698.5609012171168,"max_cost":2444.963154259909},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7-fast","name":"Anthropic: Claude Opus 4.7 (Fast)","created":1778613011,"description":"Fast-mode variant of [Opus 4.7](/anthropic/claude-opus-4.7) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.4815e-05,"completion":0.00017407500000000002,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-06,"input_cache_write":4.125e-05,"max_prompt_cost":34.815,"max_completion_cost":22.2816,"max_cost":52.640280000000004},"sats_pricing":{"prompt":0.05354975751488518,"completion":0.26774878757442594,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0050758063995151835,"input_cache_write":0.0634475799939398,"max_prompt_cost":53549.757514885176,"max_completion_cost":34271.84480952652,"max_cost":80967.2333625064},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.7-opus-fast-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"perceptron-mk1","name":"Perceptron: Perceptron Mk1","created":1778597029,"description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":1.7407500000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005704089600000001,"max_completion_cost":0.014260224000000002,"max_cost":0.0185382912},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.0026774878757442593,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.77359227123879,"max_completion_cost":21.933980678096972,"max_cost":28.514174881526063},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perceptron/perceptron-mk1-20260512","alias_ids":null,"forwarded_model_id":null},{"id":"ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","created":1778247440,"description":"Ring-2.6-1T is a 1T-parameter-scale thinking model with 63B active parameters, built for real-world agent workflows that require both strong capability and operational efficiency. It is optimized for coding agents, tool...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.703750000000001e-08,"completion":7.253125000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.022816358400000004,"max_completion_cost":0.047534080000000006,"max_cost":0.0646463488},"sats_pricing":{"prompt":0.000133874393787213,"completion":0.0011156199482267749,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5379031997575913e-05,"input_cache_write":0.0,"max_prompt_cost":35.09436908495516,"max_completion_cost":73.11326892698992,"max_cost":99.43404574070627},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ring-2.6-1t-20260508","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite","name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.925e-07,"completion":1.1550000000000002e-06,"request":0.0,"image":2.75e-07,"web_search":0.015400000000000002,"internal_reasoning":1.65e-06,"input_cache_read":2.75e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.20185088,"max_completion_cost":0.07569408000000001,"max_cost":0.26492928000000004},"sats_pricing":{"prompt":0.0002960887066383857,"completion":0.0017765322398303144,"request":0.001,"image":0.0004229838666262653,"web_search":23.687096531070857,"internal_reasoning":0.0025379031997575918,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":310.4715116520519,"max_completion_cost":116.42681686951948,"max_cost":407.4938590433182},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-chat-latest","name":"OpenAI: GPT Chat Latest","created":1778000212,"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":0.0,"max_prompt_cost":2.321,"max_completion_cost":4.45632,"max_cost":6.0346},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.0,"max_prompt_cost":3569.9838343256793,"max_completion_cost":6854.368961905303,"max_cost":9281.957969246765},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-chat-latest-20260505","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.3","name":"SpaceXAI: Grok 4.3","created":1777591821,"description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":1.4506250000000003,"max_completion_cost":2.9012500000000006,"max_cost":2.9012500000000006},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":2231.2398964535496,"max_completion_cost":4462.479792907099,"max_cost":4462.479792907099},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.3-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.1-8b","name":"IBM: Granite 4.1 8B","created":1777577071,"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.8025000000000005e-08,"completion":1.1605000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.015210905600000001,"max_cost":0.015210905600000001},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.00017849919171628395,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.459677332525305e-05,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":23.39624605663677,"max_cost":23.39624605663677},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"ibm-granite/granite-4.1-8b-20260429","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","created":1777570439,"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.7407500000000002e-06,"completion":8.70375e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.45632716800000006,"max_completion_cost":2.28163584,"max_cost":2.28163584},"sats_pricing":{"prompt":0.0026774878757442593,"completion":0.013387439378721295,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":701.8873816991031,"max_completion_cost":3509.436908495515,"max_cost":3509.436908495515},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-medium-3.5-20260430","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni","created":1777393095,"description":"PPQ.AI model","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08912640000000001,"max_completion_cost":0.29708799999999996,"max_cost":0.29708799999999996},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":137.08737923810608,"max_completion_cost":456.95793079368684,"max_cost":456.95793079368684},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-latest","name":"Anthropic Claude Haiku Latest","created":1777318492,"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":5.802500000000001e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":1.3750000000000002e-06,"max_prompt_cost":0.2321,"max_completion_cost":0.37136,"max_cost":0.529188},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.008924959585814199,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0021149193331313266,"max_prompt_cost":356.9983834325679,"max_completion_cost":571.1974134921087,"max_cost":813.9563142262548},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-haiku-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-mini-latest","name":"OpenAI GPT Mini Latest","created":1777318471,"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":8.703750000000001e-07,"completion":5.22225e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.34815000000000007,"max_completion_cost":0.6684479999999999,"max_cost":0.9051899999999999},"sats_pricing":{"prompt":0.0013387439378721297,"completion":0.008032463627232778,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00012689515998787958,"input_cache_write":0.0,"max_prompt_cost":535.497575148852,"max_completion_cost":1028.1553442857953,"max_cost":1392.2936953870146},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~openai/gpt-mini-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-pro-latest","name":"Google Gemini Pro Latest","created":1777318451,"description":"This model always redirects to the latest model in the Google Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.54e-06,"completion":9.240000000000001e-06,"request":0.0,"image":2.2e-06,"web_search":0.015400000000000002,"internal_reasoning":1.32e-05,"input_cache_read":2.2e-07,"input_cache_write":4.125e-07,"max_prompt_cost":1.61480704,"max_completion_cost":0.6055526400000001,"max_cost":2.1194342400000004},"sats_pricing":{"prompt":0.0023687096531070854,"completion":0.014212257918642515,"request":0.001,"image":0.0033838709330101225,"web_search":23.687096531070857,"internal_reasoning":0.020303225598060734,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0006344757999393979,"max_prompt_cost":2483.772093216415,"max_completion_cost":931.4145349561559,"max_cost":3259.9508723465456},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~google/gemini-pro-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-latest","name":"MoonshotAI Kimi Latest","created":1777318428,"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.6247e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.19e-07,"input_cache_write":0.0,"max_prompt_cost":3.0421811200000004,"max_completion_cost":17.036214272,"max_cost":17.036214272},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.02498988684027975,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004906612852864677,"input_cache_write":0.0,"max_prompt_cost":4679.249211327355,"max_completion_cost":26203.79558343318,"max_cost":26203.79558343318},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":1048576,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~moonshotai/kimi-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-flash-latest","name":"Google Gemini Flash Latest","created":1777318398,"description":"This model always redirects to the latest model in the Google Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":1.1550000000000002e-06,"completion":5.775e-06,"request":0.0,"image":1.65e-06,"web_search":0.015400000000000002,"internal_reasoning":8.25e-06,"input_cache_read":1.65e-07,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":1.2111052800000002,"max_completion_cost":0.3784704,"max_cost":1.5138816},"sats_pricing":{"prompt":0.0017765322398303144,"completion":0.00888266119915157,"request":0.001,"image":0.0025379031997575918,"web_search":23.687096531070857,"internal_reasoning":0.012689515998787959,"input_cache_read":0.00025379031997575915,"input_cache_write":0.00014099462220875506,"max_prompt_cost":1862.8290699123118,"max_completion_cost":582.1340843475973,"max_cost":2328.5363373903892},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~google/gemini-flash-latest","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","created":1777318368,"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":2.7500000000000004e-06,"max_prompt_cost":2.3209999999999997,"max_completion_cost":1.48544,"max_cost":3.509352},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.004229838666262653,"max_prompt_cost":3569.9838343256783,"max_completion_cost":2284.789653968435,"max_cost":5397.815557500426},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-sonnet-latest","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-latest","name":"OpenAI GPT Latest","created":1777318334,"description":"This model always redirects to the latest model in the OpenAI GPT family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":6.092625000000001,"max_completion_cost":4.45632,"max_cost":9.806225000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":9371.207565104909,"max_completion_cost":6854.368961905303,"max_cost":15083.181700025994},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~openai/gpt-latest","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","created":1777261368,"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":2.0889e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":4.125e-07,"max_prompt_cost":0.34815000000000007,"max_completion_cost":0.1368981504,"max_cost":0.4622317920000001},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.003212985450893111,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0006344757999393979,"max_prompt_cost":535.497575148852,"max_completion_cost":210.5662145097309,"max_cost":710.9694205736278},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-plus-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-flash","name":"Qwen: Qwen3.6 Flash","created":1777261362,"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.1759375000000003e-07,"completion":1.3055625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":2.578125e-07,"max_prompt_cost":0.21759375000000003,"max_completion_cost":0.085561344,"max_cost":0.28889487},"sats_pricing":{"prompt":0.0003346859844680324,"completion":0.0020081159068081945,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0003965473749621237,"max_prompt_cost":334.6859844680324,"max_completion_cost":131.60388406858183,"max_cost":444.3558878585173},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-35b-a3b","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.6247000000000002e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.042590535680000007,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.00024989886840279756,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.459677332525305e-05,"input_cache_write":0.0,"max_prompt_cost":65.50948895858296,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-max-preview","name":"Qwen: Qwen3.6 Max Preview","created":1777260242,"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.1918335000000001e-06,"completion":7.151001000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":1.412125e-06,"max_prompt_cost":0.31243200102400004,"max_completion_cost":0.46864800153600006,"max_cost":0.702972002304},"sats_pricing":{"prompt":0.0018331866989262362,"completion":0.010999120193557418,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0021720221551258722,"max_prompt_cost":480.55889400331927,"max_completion_cost":720.8383410049789,"max_cost":1081.2575115074683},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-max-preview-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-27b","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":4.1778e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.32e-07,"input_cache_write":0.0,"max_prompt_cost":0.18253086720000003,"max_completion_cost":1.0951852032,"max_cost":1.0951852032},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.006425970901786222,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002030322559806073,"input_cache_write":0.0,"max_prompt_cost":280.7549526796413,"max_completion_cost":1684.5297160778473,"max_cost":1684.5297160778473},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-27b-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":0.0,"max_prompt_cost":6.092625000000001,"max_completion_cost":4.45632,"max_cost":9.806225000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.0,"max_prompt_cost":9371.207565104909,"max_completion_cost":6854.368961905303,"max_cost":15083.181700025994},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-1t","name":"inclusionAI: Ling-2.6-1T","created":1776948238,"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.703750000000001e-08,"completion":7.253125000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.022816358400000004,"max_completion_cost":0.023767040000000003,"max_cost":0.04373135360000001},"sats_pricing":{"prompt":0.000133874393787213,"completion":0.0011156199482267749,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5379031997575913e-05,"input_cache_write":0.0,"max_prompt_cost":35.09436908495516,"max_completion_cost":36.55663446349496,"max_cost":67.26420741283073},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ling-2.6-1t-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"hy3-preview","name":"Tencent: Hy3 preview","created":1776878150,"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.31115e-08,"completion":2.43705e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.3100000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.019165741056,"max_completion_cost":0.06388580352,"max_cost":0.06388580352},"sats_pricing":{"prompt":0.00011245449078125889,"completion":0.0003748483026041963,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.5530644796606286e-05,"input_cache_write":0.0,"max_prompt_cost":29.47927003136233,"max_completion_cost":98.26423343787444,"max_cost":98.26423343787444},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"tencent/hy3-preview-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5-pro","name":"Xiaomi: MiMo-V2.5-Pro","created":1776874273,"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.048175000000001e-07,"completion":1.0096350000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.96e-09,"input_cache_write":0.0,"max_prompt_cost":0.5293395148800001,"max_completion_cost":0.13233487872000002,"max_cost":0.5955069542400001},"sats_pricing":{"prompt":0.0007764714839658352,"completion":0.0015529429679316704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.090967679418221e-06,"input_cache_write":0.0,"max_prompt_cost":814.1893627709596,"max_completion_cost":203.5473406927399,"max_cost":915.9630331173297},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"mimo-v2.5","name":"Xiaomi: MiMo-V2.5","created":1776874269,"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.6247000000000002e-07,"completion":3.2494000000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.08e-09,"input_cache_write":0.0,"max_prompt_cost":0.17036214272000003,"max_completion_cost":0.042590535680000007,"max_cost":0.19165741056000005},"sats_pricing":{"prompt":0.00024989886840279756,"completion":0.0004997977368055951,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.7374193062141715e-06,"input_cache_write":0.0,"max_prompt_cost":262.03795583433185,"max_completion_cost":65.50948895858296,"max_cost":294.79270031362336},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"xiaomi/mimo-v2.5-20260422","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","created":1776797528,"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.284e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.2e-06,"input_cache_write":0.0,"max_prompt_cost":2.525248,"max_completion_cost":2.22816,"max_cost":3.565056},"sats_pricing":{"prompt":0.014279935337302714,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0033838709330101225,"input_cache_write":0.0,"max_prompt_cost":3884.1424117463384,"max_completion_cost":3427.1844809526515,"max_cost":5483.495169524243},"per_request_limits":null,"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-image-2-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"ling-2.6-flash","name":"inclusionAI: Ling-2.6-flash","created":1776795886,"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605e-08,"completion":3.4815e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000003e-09,"input_cache_write":0.0,"max_prompt_cost":0.00304218112,"max_completion_cost":0.00114081792,"max_cost":0.0038027264},"sats_pricing":{"prompt":1.7849919171628396e-05,"completion":5.354975751488518e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3838709330101227e-06,"input_cache_write":0.0,"max_prompt_cost":4.679249211327354,"max_completion_cost":1.7547184542477576,"max_cost":5.849061514159192},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inclusionai/ling-2.6-flash-20260421","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-latest","name":"Anthropic: Claude Opus Latest","created":1776795361,"description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":2.9012500000000004e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.802500000000001,"max_completion_cost":3.7136000000000005,"max_cost":8.773380000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.044624797929070995,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":8924.959585814198,"max_completion_cost":5711.974134921087,"max_cost":13494.538893751069},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"~anthropic/claude-opus-latest","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.7250975e-07,"completion":2.8316200000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0736e-07,"input_cache_write":0.0,"max_prompt_cost":0.176294395904,"max_completion_cost":0.7422921932800001,"max_cost":0.7422921932800001},"sats_pricing":{"prompt":0.0010344028159958655,"completion":0.004355380277877329,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016513290153089395,"input_cache_write":0.0,"max_prompt_cost":271.16249179642017,"max_completion_cost":1141.7368075638744,"max_cost":1141.7368075638744},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2.6-20260420","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7","name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":2.9012500000000004e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.802500000000001,"max_completion_cost":3.7136000000000005,"max_cost":8.773380000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.044624797929070995,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":8924.959585814198,"max_completion_cost":5711.974134921087,"max_cost":13494.538893751069},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.1","name":"Z.ai: GLM 5.1","created":1775578025,"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1047959999999999e-06,"completion":3.472216e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.9448e-07,"input_cache_write":0.0,"max_prompt_cost":0.22399959859199997,"max_completion_cost":0.455110295552,"max_cost":0.534302072832},"sats_pricing":{"prompt":0.001699312305139023,"completion":0.0053406958161512155,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0002991341904780948,"input_cache_write":0.0,"max_prompt_cost":344.53896849154717,"max_completion_cost":700.0156820145721,"max_cost":821.8223880469374},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5.1-20260406","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-26b-a4b-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":8.123500000000001e-08,"completion":3.9457e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021295267840000003,"max_completion_cost":0.00646463488,"max_cost":0.026428948480000002},"sats_pricing":{"prompt":0.00012494943420139878,"completion":0.0006068972518353654,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.75474447929148,"max_completion_cost":9.943404574070627,"max_cost":40.65097752340639},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-4-31b-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":3.9457e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.030421811200000003,"max_completion_cost":0.10343415808,"max_cost":0.10343415808},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0006068972518353654,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":46.79249211327354,"max_completion_cost":159.09447318513003,"max_cost":159.09447318513003},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-4-31b-it-20260402","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.6-plus","name":"Qwen: Qwen3.6 Plus","created":1775133557,"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.771625e-07,"completion":2.2629749999999997e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":4.46875e-07,"max_prompt_cost":0.3771625,"max_completion_cost":0.14830632959999998,"max_cost":0.500751108},"sats_pricing":{"prompt":0.0005801223730779228,"completion":0.0034807342384675366,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.000687348783267681,"max_prompt_cost":580.1223730779228,"max_completion_cost":228.11339905220848,"max_cost":770.2168722880965},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.6-plus-04-02","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5v-turbo","name":"Z.ai: GLM 5V Turbo","created":1775061458,"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3926000000000002e-06,"completion":4.642e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.64e-07,"input_cache_write":0.0,"max_prompt_cost":0.28235243520000003,"max_completion_cost":0.608436224,"max_cost":0.7082577919999999},"sats_pricing":{"prompt":0.002141990300595408,"completion":0.007139967668651357,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004060645119612146,"input_cache_write":0.0,"max_prompt_cost":434.2928174263201,"max_completion_cost":935.8498422654707,"max_cost":1089.3877070121493},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5v-turbo-20260401","alias_ids":null,"forwarded_model_id":null},{"id":"trinity-large-thinking","name":"Arcee AI: Trinity Large Thinking","created":1775058318,"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.5531000000000003e-07,"completion":9.86425e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-08,"input_cache_write":0.0,"max_prompt_cost":0.06692798464000001,"max_completion_cost":0.2585853952,"max_cost":0.2585853952},"sats_pricing":{"prompt":0.00039269822177582473,"completion":0.0015172431295884135,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010151612799030365,"input_cache_write":0.0,"max_prompt_cost":102.9434826492018,"max_completion_cost":397.73618296282507,"max_cost":397.73618296282507},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"arcee-ai/trinity-large-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","created":1774979158,"description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":2.9012500000000006,"max_completion_cost":5.802500000000001,"max_cost":5.802500000000001},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":4462.479792907099,"max_completion_cost":8924.959585814198,"max_cost":8924.959585814198},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"grok-4.20","name":"SpaceXAI: Grok 4.20","created":1774979019,"description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":2.9012500000000006,"max_completion_cost":5.802500000000001,"max_cost":5.802500000000001},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":4462.479792907099,"max_completion_cost":8924.959585814198,"max_cost":8924.959585814198},"per_request_limits":null,"top_provider":{"context_length":2000000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"x-ai/grok-4.20-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"lyria-3-pro-preview","name":"Lyria 3 Pro Preview","created":1774907286,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"lyria-3-clip-preview","name":"Lyria 3 Clip Preview","created":1774907255,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":0.0,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.0},"sats_pricing":{"prompt":0.0,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0,"max_completion_cost":0.0,"max_cost":0.001},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","created":1774649310,"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-08,"input_cache_write":0.0,"max_prompt_cost":0.08912640000000001,"max_completion_cost":0.11140800000000002,"max_cost":0.1726824},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010151612799030365,"input_cache_write":0.0,"max_prompt_cost":137.08737923810608,"max_completion_cost":171.3592240476326,"max_cost":265.6067972738305},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":80000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","alias_ids":null,"forwarded_model_id":null},{"id":"reka-edge","name":"Reka Edge","created":1774026965,"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":1.1605000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0019013632000000002,"max_completion_cost":0.0019013632000000002,"max_cost":0.0019013632000000002},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.00017849919171628395,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.924530757079596,"max_completion_cost":2.924530757079596,"max_cost":2.924530757079596},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"rekaai/reka-edge-2603","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.7","name":"MiniMax: MiniMax M2.7","created":1773836697,"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.1333500000000005e-07,"completion":1.2533400000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.6e-08,"input_cache_write":0.0,"max_prompt_cost":0.06417100800000002,"max_completion_cost":0.16427778048000002,"max_cost":0.18737934336000003},"sats_pricing":{"prompt":0.0004819478176339667,"completion":0.0019277912705358668,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010151612799030365,"input_cache_write":0.0,"max_prompt_cost":98.7029130514364,"max_completion_cost":252.67945741167713,"max_cost":288.2125061101942},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2.7-20260318","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.18253086720000003,"max_cost":0.18253086720000003},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5379031997575913e-05,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":280.7549526796413,"max_cost":280.7549526796413},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-2603","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-turbo","name":"Z.ai: GLM 5 Turbo","created":1773583573,"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.3926000000000002e-06,"completion":4.642e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.64e-07,"input_cache_write":0.0,"max_prompt_cost":0.28235243520000003,"max_completion_cost":0.608436224,"max_cost":0.7082577919999999},"sats_pricing":{"prompt":0.002141990300595408,"completion":0.007139967668651357,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004060645119612146,"input_cache_write":0.0,"max_prompt_cost":434.2928174263201,"max_completion_cost":935.8498422654707,"max_cost":1089.3877070121493},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5-turbo-20260315","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-super-120b-a12b","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.86425e-08,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02585853952,"max_completion_cost":0.007605452800000001,"max_cost":0.0318478336},"sats_pricing":{"prompt":0.00015172431295884135,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":39.77361829628251,"max_completion_cost":11.698123028318385,"max_cost":48.985890181083235},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-lite","name":"ByteDance Seed: Seed-2.0-Lite","created":1773157231,"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.076054528,"max_completion_cost":0.304218112,"max_cost":0.342245376},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":116.98123028318383,"max_completion_cost":467.92492113273534,"max_cost":526.4155362743273},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-9b","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":1.7407500000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030421811200000003,"max_completion_cost":0.04563271680000001,"max_cost":0.04563271680000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.000267748787574426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.79249211327354,"max_completion_cost":70.18873816991032,"max_cost":70.18873816991032},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-9b-20260310","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro","name":"OpenAI: GPT-5.4 Pro","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.4815e-05,"completion":0.00020889000000000001,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-06,"input_cache_write":0.0,"max_prompt_cost":36.555749999999996,"max_completion_cost":26.737920000000003,"max_cost":58.83735},"sats_pricing":{"prompt":0.05354975751488518,"completion":0.3212985450893111,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0050758063995151835,"input_cache_write":0.0,"max_prompt_cost":56227.24539062943,"max_completion_cost":41126.21377143182,"max_cost":90499.09020015596},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":0.0,"max_prompt_cost":3.0463125000000004,"max_completion_cost":2.22816,"max_cost":4.903112500000001},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0004229838666262653,"input_cache_write":0.0,"max_prompt_cost":4685.6037825524545,"max_completion_cost":3427.1844809526515,"max_cost":7541.590850012997},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"mercury-2","name":"Inception: Mercury 2","created":1772636275,"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":8.703750000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.037135999999999995,"max_completion_cost":0.04351875000000001,"max_cost":0.06614850000000001},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0013387439378721297,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.0,"max_prompt_cost":57.119741349210855,"max_completion_cost":66.9371968936065,"max_cost":101.74453927828186},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"inception/mercury-2-20260304","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite-preview","name":"Google: Gemini 3.1 Flash Lite Preview","created":1772512673,"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.925e-07,"completion":1.1550000000000002e-06,"request":0.0,"image":2.75e-07,"web_search":0.015400000000000002,"internal_reasoning":1.65e-06,"input_cache_read":2.75e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.20185088,"max_completion_cost":0.07569408000000001,"max_cost":0.26492928000000004},"sats_pricing":{"prompt":0.0002960887066383857,"completion":0.0017765322398303144,"request":0.001,"image":0.0004229838666262653,"web_search":23.687096531070857,"internal_reasoning":0.0025379031997575918,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":310.4715116520519,"max_completion_cost":116.42681686951948,"max_cost":407.4938590433182},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","alias_ids":null,"forwarded_model_id":null},{"id":"seed-2.0-mini","name":"ByteDance Seed: Seed-2.0-Mini","created":1772131107,"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030421811200000003,"max_completion_cost":0.060843622400000005,"max_cost":0.07605452800000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.79249211327354,"max_completion_cost":93.58498422654708,"max_cost":116.98123028318386},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-image-preview","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","created":1772119558,"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.85e-07,"completion":2.3100000000000003e-06,"request":0.0,"image":0.0,"web_search":0.015400000000000002,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02523136,"max_completion_cost":0.15138816000000002,"max_cost":0.15138816000000002},"sats_pricing":{"prompt":0.0005921774132767714,"completion":0.003553064479660629,"request":0.001,"image":0.0,"web_search":23.687096531070857,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.80893895650649,"max_completion_cost":232.85363373903897,"max_cost":232.85363373903897},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-35b-a3b","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.6247000000000002e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.042590535680000007,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.00024989886840279756,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":65.50948895858296,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-27b","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2629750000000002e-07,"completion":1.8103800000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.059322531840000005,"max_completion_cost":0.11864506368000001,"max_cost":0.16313696256000002},"sats_pricing":{"prompt":0.0003480734238467537,"completion":0.0027845873907740297,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":91.2453596208834,"max_completion_cost":182.4907192417668,"max_cost":250.92473895742935},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-27b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-122b-a10b","name":"Qwen: Qwen3.5-122B-A10B","created":1772053789,"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.3654500000000005e-07,"completion":2.7852000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08822325248000001,"max_completion_cost":0.22816358400000003,"max_cost":0.28881707008},"sats_pricing":{"prompt":0.0005176476559772235,"completion":0.004283980601190816,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":135.69822712849327,"max_completion_cost":350.94369084955156,"max_cost":444.2362220003907},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","created":1772053776,"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.543250000000001e-08,"completion":3.0173000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07543250000000001,"max_completion_cost":0.019774177280000003,"max_cost":0.09026313296},"sats_pricing":{"prompt":0.00011602447461558457,"completion":0.0004640978984623383,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":116.02447461558458,"max_completion_cost":30.415119873627802,"max_cost":138.83581452080543},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-flash-20260224","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview-customtools","name":"Google: Gemini 3.1 Pro Preview Custom Tools","created":1772045923,"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.54e-06,"completion":9.240000000000001e-06,"request":0.0,"image":2.2e-06,"web_search":0.015400000000000002,"internal_reasoning":1.32e-05,"input_cache_read":2.2e-07,"input_cache_write":4.125e-07,"max_prompt_cost":1.61480704,"max_completion_cost":0.6055526400000001,"max_cost":2.1194342400000004},"sats_pricing":{"prompt":0.0023687096531070854,"completion":0.014212257918642515,"request":0.001,"image":0.0033838709330101225,"web_search":23.687096531070857,"internal_reasoning":0.020303225598060734,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0006344757999393979,"max_prompt_cost":2483.772093216415,"max_completion_cost":931.4145349561559,"max_cost":3259.9508723465456},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"aion-2.0","name":"AionLabs: Aion-2.0","created":1771881306,"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.284000000000001e-07,"completion":1.8568000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.12168724480000001,"max_completion_cost":0.060843622400000005,"max_cost":0.15210905600000002},"sats_pricing":{"prompt":0.0014279935337302716,"completion":0.002855987067460543,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":187.16996845309416,"max_completion_cost":93.58498422654708,"max_cost":233.96246056636772},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-2.0-20260223","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview","name":"Google: Gemini 3.1 Pro Preview","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.54e-06,"completion":9.240000000000001e-06,"request":0.0,"image":2.2e-06,"web_search":0.015400000000000002,"internal_reasoning":1.32e-05,"input_cache_read":2.2e-07,"input_cache_write":4.125e-07,"max_prompt_cost":1.61480704,"max_completion_cost":0.6055526400000001,"max_cost":2.1194342400000004},"sats_pricing":{"prompt":0.0023687096531070854,"completion":0.014212257918642515,"request":0.001,"image":0.0033838709330101225,"web_search":23.687096531070857,"internal_reasoning":0.020303225598060734,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0006344757999393979,"max_prompt_cost":2483.772093216415,"max_completion_cost":931.4145349561559,"max_cost":3259.9508723465456},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-07,"input_cache_write":4.125e-06,"max_prompt_cost":3.4815000000000005,"max_completion_cost":2.22816,"max_cost":5.264028},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0005075806399515183,"input_cache_write":0.006344757999393979,"max_prompt_cost":5354.975751488519,"max_completion_cost":3427.1844809526515,"max_cost":8096.7233362506395},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","created":1771229416,"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.0173000000000004e-07,"completion":1.8103800000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.30173000000000005,"max_completion_cost":0.11864506368000001,"max_cost":0.4006008864},"sats_pricing":{"prompt":0.0004640978984623383,"completion":0.0027845873907740297,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":464.0978984623383,"max_completion_cost":182.4907192417668,"max_cost":616.1734978304772},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-plus-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3.5-397b-a17b","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.5259500000000004e-07,"completion":2.7155700000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11864506368000001,"max_completion_cost":0.17796759552000002,"max_cost":0.26695139328000006},"sats_pricing":{"prompt":0.0006961468476935074,"completion":0.004176881086161045,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":182.4907192417668,"max_completion_cost":273.73607886265023,"max_cost":410.6041182939754},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.5","name":"MiniMax: MiniMax M2.5","created":1770908502,"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.5531000000000003e-07,"completion":1.04445e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":4.125e-07,"max_prompt_cost":0.05019598848000001,"max_completion_cost":0.20534722560000002,"max_cost":0.20534722560000002},"sats_pricing":{"prompt":0.00039269822177582473,"completion":0.0016064927254465554,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.459677332525305e-05,"input_cache_write":0.0006344757999393979,"max_prompt_cost":77.20761198690136,"max_completion_cost":315.8493217645964,"max_cost":315.8493217645964},"per_request_limits":null,"top_provider":{"context_length":196608,"max_completion_tokens":196608,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2.5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5","name":"Z.ai: GLM 5","created":1770829182,"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1024750000000002e-06,"completion":2.959275e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.22578688000000005,"max_completion_cost":0.3878780928,"max_cost":0.46916136960000004},"sats_pricing":{"prompt":0.0016957423213046978,"completion":0.0045517293887652405,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":347.28802740320214,"max_completion_cost":596.6042744442376,"max_cost":721.6279643093904},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-5-20260211","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","created":1770671901,"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":9.051900000000001e-07,"completion":4.525949999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.23729012736000002,"max_completion_cost":0.29661265919999996,"max_cost":0.47458025472},"sats_pricing":{"prompt":0.0013922936953870149,"completion":0.006961468476935073,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":364.9814384835336,"max_completion_cost":456.22679810441696,"max_cost":729.9628769670671},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-max-thinking-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6","name":"Anthropic: Claude Opus 4.6","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":2.9012500000000004e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":5.802500000000001,"max_completion_cost":3.7136000000000005,"max_cost":8.773380000000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.044624797929070995,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":8924.959585814198,"max_completion_cost":5711.974134921087,"max_cost":13494.538893751069},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-next","name":"Qwen: Qwen3 Coder Next","created":1770164101,"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.3926e-07,"completion":9.284000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.700000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.03650617344,"max_completion_cost":0.24337448960000002,"max_cost":0.24337448960000002},"sats_pricing":{"prompt":0.00021419903005954073,"completion":0.0014279935337302716,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001184354826553543,"input_cache_write":0.0,"max_prompt_cost":56.150990535928244,"max_completion_cost":374.3399369061883,"max_cost":374.3399369061883},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","alias_ids":null,"forwarded_model_id":null},{"id":"step-3.5-flash","name":"StepFun: Step 3.5 Flash","created":1769728337,"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":3.4815000000000006e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030421811200000003,"max_completion_cost":0.022816358400000004,"max_cost":0.0456327168},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.000535497575148852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.79249211327354,"max_completion_cost":35.09436908495516,"max_cost":70.18873816991031},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"stepfun/step-3.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.614850000000002e-07,"completion":3.307425e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.0450000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.17340432384000004,"max_completion_cost":0.8670216192,"max_cost":0.8670216192},"sats_pricing":{"prompt":0.0010174453927828187,"completion":0.005087226963914092,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00016073386931798084,"input_cache_write":0.0,"max_prompt_cost":266.7172050456592,"max_completion_cost":1333.5860252282957,"max_cost":1333.5860252282957},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2.5-0127","alias_ids":null,"forwarded_model_id":null},{"id":"solar-pro-3","name":"Upstage: Solar Pro 3","created":1769481200,"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.022816358400000004,"max_completion_cost":0.09126543360000001,"max_cost":0.09126543360000001},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5379031997575913e-05,"input_cache_write":0.0,"max_prompt_cost":35.09436908495516,"max_completion_cost":140.37747633982065,"max_cost":140.37747633982065},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"upstage/solar-pro-3","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2-her","name":"MiniMax: MiniMax M2-her","created":1769177239,"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.022816358400000004,"max_completion_cost":0.0028520448000000005,"max_cost":0.024955392000000007},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":35.09436908495516,"max_completion_cost":4.386796135619395,"max_cost":38.38446618666971},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2-her-20260123","alias_ids":null,"forwarded_model_id":null},{"id":"palmyra-x5","name":"Writer: Palmyra X5","created":1769003823,"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.7241520000000001,"max_completion_cost":0.05704089600000001,"max_cost":0.7754888064000001},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1113.834956309612,"max_completion_cost":87.73592271238789,"max_cost":1192.797286750761},"per_request_limits":null,"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"writer/palmyra-x5-20250428","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio","name":"OpenAI: GPT Audio","created":1768862569,"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.37136,"max_completion_cost":0.19013632000000003,"max_cost":0.5139622400000001},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":571.1974134921087,"max_completion_cost":292.45307570795967,"max_cost":790.5372202730786},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-audio","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-audio-mini","name":"OpenAI: GPT Audio Mini","created":1768859419,"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":2.7852000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08912640000000001,"max_completion_cost":0.04563271680000001,"max_cost":0.12335093760000002},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.004283980601190816,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":137.08737923810608,"max_completion_cost":70.18873816991032,"max_cost":189.7289328655388},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-audio-mini","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7-flash","name":"Z.ai: GLM 4.7 Flash","created":1768833913,"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963e-08,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.014117621759999999,"max_completion_cost":0.007605452800000001,"max_cost":0.020582256639999998},"sats_pricing":{"prompt":0.00010709951502977036,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6919354665050612e-05,"input_cache_write":0.0,"max_prompt_cost":21.714640871316,"max_completion_cost":11.698123028318385,"max_cost":31.658045445386623},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.7-flash-20260119","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-codex","name":"OpenAI: GPT-5.2-Codex","created":1768409315,"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.030875e-06,"completion":1.6247e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.925e-07,"input_cache_write":0.0,"max_prompt_cost":0.8123499999999999,"max_completion_cost":2.0796159999999997,"max_cost":2.632014},"sats_pricing":{"prompt":0.003123735855034969,"completion":0.02498988684027975,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0002960887066383857,"input_cache_write":0.0,"max_prompt_cost":1249.4943420139875,"max_completion_cost":3198.705515555808,"max_cost":4048.3616681253197},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-codex-20260114","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","created":1766505011,"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.703750000000001e-08,"completion":3.4815000000000006e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.022816358400000004,"max_completion_cost":0.011408179200000002,"max_cost":0.031372492800000006},"sats_pricing":{"prompt":0.000133874393787213,"completion":0.000535497575148852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":35.09436908495516,"max_completion_cost":17.54718454247758,"max_cost":48.25475749181334},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"seed-1.6","name":"ByteDance Seed: Seed 1.6","created":1766504997,"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.076054528,"max_completion_cost":0.076054528,"max_cost":0.14260224},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":116.98123028318383,"max_completion_cost":116.98123028318383,"max_cost":219.3398067809697},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance-seed/seed-1.6-20250625","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2.1","name":"MiniMax: MiniMax M2.1","created":1766454997,"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":4.125e-07,"max_prompt_cost":0.07130112000000001,"max_completion_cost":0.18253086720000003,"max_cost":0.20819927040000002},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0006344757999393979,"max_prompt_cost":109.66990339048486,"max_completion_cost":280.7549526796413,"max_cost":320.2361179002158},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":2.030875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.800000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.09411747840000001,"max_completion_cost":0.266190848,"max_cost":0.29946470399999997},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.003123735855034969,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001353548373204049,"input_cache_write":0.0,"max_prompt_cost":144.76427247544,"max_completion_cost":409.43430599114345,"max_cost":460.6135942400363},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.7-20251222","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-nano-30b-a3b","name":"NVIDIA: Nemotron 3 Nano 30B A3B","created":1765731275,"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.8025000000000005e-08,"completion":2.3210000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.015210905600000001,"max_completion_cost":0.060843622400000005,"max_cost":0.060843622400000005},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.0003569983834325679,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":23.39624605663677,"max_completion_cost":93.58498422654708,"max_cost":93.58498422654708},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","created":1765389783,"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.030875e-06,"completion":1.6247e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.925e-07,"input_cache_write":0.0,"max_prompt_cost":0.25995199999999996,"max_completion_cost":0.266190848,"max_cost":0.49286899199999995},"sats_pricing":{"prompt":0.003123735855034969,"completion":0.02498988684027975,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0002960887066383857,"input_cache_write":0.0,"max_prompt_cost":399.838189444476,"max_completion_cost":409.43430599114345,"max_cost":758.0932071867264},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-chat-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro","name":"OpenAI: GPT-5.2 Pro","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.4370500000000003e-05,"completion":0.00019496400000000003,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.7482,"max_completion_cost":24.955392000000003,"max_cost":31.584168000000005},"sats_pricing":{"prompt":0.037484830260419634,"completion":0.29987864208335707,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14993.932104167852,"max_completion_cost":38384.466186669706,"max_cost":48580.34001750385},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2","name":"OpenAI: GPT-5.2","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.030875e-06,"completion":1.6247e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.925e-07,"input_cache_write":0.0,"max_prompt_cost":0.8123499999999999,"max_completion_cost":2.0796159999999997,"max_cost":2.632014},"sats_pricing":{"prompt":0.003123735855034969,"completion":0.02498988684027975,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0002960887066383857,"input_cache_write":0.0,"max_prompt_cost":1249.4943420139875,"max_completion_cost":3198.705515555808,"max_cost":4048.3616681253197},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"relace-search","name":"Relace: Relace Search","created":1765213560,"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":3.4815000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.29708799999999996,"max_completion_cost":0.4456320000000001,"max_cost":0.594176},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.005354975751488519,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":456.95793079368684,"max_completion_cost":685.4368961905304,"max_cost":913.9158615873738},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"relace/relace-search-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6v","name":"Z.ai: GLM 4.6V","created":1765207462,"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.04445e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.05e-08,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.0342245376,"max_cost":0.06844907520000001},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0016064927254465554,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.305645065777837e-05,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":52.64155362743273,"max_cost":105.28310725486548},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.6-20251208","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-max","name":"OpenAI: GPT-5.1-Codex-Max","created":1764878934,"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":0.0,"max_prompt_cost":0.58025,"max_completion_cost":1.48544,"max_cost":1.8800100000000002},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00021149193331313266,"input_cache_write":0.0,"max_prompt_cost":892.4959585814198,"max_completion_cost":2284.789653968435,"max_cost":2891.6869058038},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-codex-max-20251204","alias_ids":null,"forwarded_model_id":null},{"id":"nova-2-lite-v1","name":"Amazon: Nova 2 Lite","created":1764696672,"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.34815000000000007,"max_completion_cost":0.19013341875000003,"max_cost":0.5154674085000001},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":535.497575148852,"max_completion_cost":292.4486132281668,"max_cost":792.8523547896386},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-2-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","created":1764681735,"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":2.3210000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.060843622400000005,"max_completion_cost":0.060843622400000005,"max_cost":0.060843622400000005},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.0003569983834325679,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3838709330101225e-05,"input_cache_write":0.0,"max_prompt_cost":93.58498422654708,"max_completion_cost":93.58498422654708,"max_cost":93.58498422654708},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/ministral-14b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","created":1764681654,"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":1.7407500000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-08,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.04563271680000001,"max_cost":0.04563271680000001},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.000267748787574426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.5379031997575913e-05,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":70.18873816991032,"max_cost":70.18873816991032},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/ministral-8b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","created":1764681560,"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":1.1605000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.015210905600000001,"max_completion_cost":0.015210905600000001,"max_cost":0.015210905600000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.00017849919171628395,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6919354665050612e-05,"input_cache_write":0.0,"max_prompt_cost":23.39624605663677,"max_completion_cost":23.39624605663677,"max_cost":23.39624605663677},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/ministral-3b-2512","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2512","name":"Mistral: Mistral Large 3 2512","created":1764624472,"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.8025e-07,"completion":1.7407500000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.152109056,"max_completion_cost":0.45632716800000006,"max_cost":0.45632716800000006},"sats_pricing":{"prompt":0.0008924959585814196,"completion":0.0026774878757442593,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.459677332525305e-05,"input_cache_write":0.0,"max_prompt_cost":233.96246056636767,"max_completion_cost":701.8873816991031,"max_cost":701.8873816991031},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-large-2512","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":3.0173000000000004e-07,"completion":4.4099000000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4795e-07,"input_cache_write":0.0,"max_prompt_cost":0.049435443200000005,"max_completion_cost":0.028900720640000002,"max_cost":0.05856198656000001},"sats_pricing":{"prompt":0.0004640978984623383,"completion":0.0006782969285218791,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00022756532024493072,"input_cache_write":0.0,"max_prompt_cost":76.03779968406951,"max_completion_cost":44.45286750760987,"max_cost":90.07554731805158},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v3.2-20251201","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5","name":"Anthropic: Claude Opus 4.5","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":2.9012500000000004e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":6.875000000000001e-06,"max_prompt_cost":1.1605,"max_completion_cost":1.8568000000000002,"max_cost":2.6459400000000004},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.044624797929070995,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.010574596665656633,"max_prompt_cost":1784.9919171628396,"max_completion_cost":2855.9870674605436,"max_cost":4069.7815711312746},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"olmo-3-32b-think","name":"AllenAI: Olmo 3 32B Think","created":1763758276,"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":5.8025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.011408179200000002,"max_completion_cost":0.038027264,"max_cost":0.038027264},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.0008924959585814196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":17.54718454247758,"max_completion_cost":58.49061514159192,"max_cost":58.49061514159192},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"allenai/olmo-3-32b-think-20251121","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-pro-image-preview","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","created":1763653797,"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.54e-06,"completion":9.240000000000001e-06,"request":0.0,"image":2.2e-06,"web_search":0.015400000000000002,"internal_reasoning":1.32e-05,"input_cache_read":2.2e-07,"input_cache_write":4.125e-07,"max_prompt_cost":0.10092544,"max_completion_cost":0.30277632000000004,"max_cost":0.35323904000000006},"sats_pricing":{"prompt":0.0023687096531070854,"completion":0.014212257918642515,"request":0.001,"image":0.0033838709330101225,"web_search":23.687096531070857,"internal_reasoning":0.020303225598060734,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0006344757999393979,"max_prompt_cost":155.23575582602595,"max_completion_cost":465.70726747807794,"max_cost":543.3251453910909},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-3-pro-image-preview-20251120","alias_ids":null,"forwarded_model_id":null},{"id":"cogito-v2.1-671b","name":"Deep Cogito: Cogito v2.1 671B","created":1763071233,"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":1.4506250000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.18568,"max_completion_cost":0.18568,"max_cost":0.18568},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.0022312398964535497,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":285.59870674605435,"max_completion_cost":285.59870674605435,"max_cost":285.59870674605435},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1","name":"OpenAI: GPT-5.1","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":0.0,"max_prompt_cost":0.58025,"max_completion_cost":1.48544,"max_cost":1.8800100000000002},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00021149193331313266,"input_cache_write":0.0,"max_prompt_cost":892.4959585814198,"max_completion_cost":2284.789653968435,"max_cost":2891.6869058038},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1-Codex","created":1763060298,"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.4300000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.58025,"max_completion_cost":1.48544,"max_cost":1.8800100000000002},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00021995161064565798,"input_cache_write":0.0,"max_prompt_cost":892.4959585814198,"max_completion_cost":2284.789653968435,"max_cost":2891.6869058038},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-codex-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1-Codex-Mini","created":1763057820,"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.11605,"max_completion_cost":0.29708799999999996,"max_cost":0.37600199999999995},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":178.49919171628395,"max_completion_cost":456.95793079368684,"max_cost":578.3373811607598},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-thinking","name":"MoonshotAI: Kimi K2 Thinking","created":1762440622,"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":0.18253086720000003,"max_completion_cost":0.29114624000000006,"max_cost":0.40380200960000007},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00025379031997575915,"input_cache_write":0.0,"max_prompt_cost":280.7549526796413,"max_completion_cost":447.81877217781323,"max_cost":621.0972195347794},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","alias_ids":null,"forwarded_model_id":null},{"id":"nova-premier-v1","name":"Amazon: Nova Premier 1.0","created":1761950332,"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.4506250000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.875000000000001e-07,"input_cache_write":0.0,"max_prompt_cost":2.9012500000000006,"max_completion_cost":0.46420000000000006,"max_cost":3.2726100000000002},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.022312398964535497,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0010574596665656633,"input_cache_write":0.0,"max_prompt_cost":4462.479792907099,"max_completion_cost":713.9967668651359,"max_cost":5033.6772063992075},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-premier-v1","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro-search","name":"Perplexity: Sonar Pro Search","created":1761854366,"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.0198,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6963000000000001,"max_completion_cost":0.13926,"max_cost":0.8077080000000001},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":30.454838397091102,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1070.995150297704,"max_completion_cost":214.19903005954072,"max_cost":1242.3543743453363},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-pro-search","alias_ids":null,"forwarded_model_id":null},{"id":"voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","created":1761835144,"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32000,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":3.4815000000000006e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1000000000000001e-08,"input_cache_write":0.0,"max_prompt_cost":0.0037136000000000005,"max_completion_cost":0.011140800000000001,"max_cost":0.011140800000000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.000535497575148852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.6919354665050612e-05,"input_cache_write":0.0,"max_prompt_cost":5.711974134921087,"max_completion_cost":17.13592240476326,"max_cost":17.13592240476326},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/voxtral-small-24b-2507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","created":1761752836,"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":8.703750000000001e-08,"completion":3.4815000000000006e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.125e-08,"input_cache_write":0.0,"max_prompt_cost":0.011408179200000002,"max_completion_cost":0.022816358400000004,"max_cost":0.028520448000000004},"sats_pricing":{"prompt":0.000133874393787213,"completion":0.000535497575148852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.344757999393979e-05,"input_cache_write":0.0,"max_prompt_cost":17.54718454247758,"max_completion_cost":35.09436908495516,"max_cost":43.867961356193945},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-oss-safeguard-20b","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-nano-12b-v2-vl","name":"Nemotron Nano 12B 2 VL","created":1761675565,"description":"PPQ.AI model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.044563200000000004,"max_completion_cost":0.14854399999999998,"max_cost":0.14854399999999998},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":68.54368961905304,"max_completion_cost":228.47896539684342,"max_cost":228.47896539684342},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m2","name":"MiniMax: MiniMax M2","created":1761252093,"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.959275e-07,"completion":1.18371e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":4.125e-07,"max_prompt_cost":0.060605952000000005,"max_completion_cost":0.15515123712,"max_cost":0.17696937984},"sats_pricing":{"prompt":0.00045517293887652405,"completion":0.0018206917555060962,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0006344757999393979,"max_prompt_cost":93.21941788191214,"max_completion_cost":238.64170977769504,"max_cost":272.2007002151834},"per_request_limits":null,"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m2","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":1.20692e-07,"completion":4.82768e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.015819341824,"max_completion_cost":0.015819341824,"max_cost":0.027683848192000003},"sats_pricing":{"prompt":0.00018563915938493532,"completion":0.0007425566375397413,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":24.332095898902242,"max_completion_cost":24.332095898902242,"max_cost":42.58116782307892},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","created":1760927695,"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.97285e-08,"completion":1.2997600000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0025844335,"max_completion_cost":0.017026856000000003,"max_cost":0.017026856000000003},"sats_pricing":{"prompt":3.034486259176827e-05,"completion":0.00019991909472223806,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.975176999521643,"max_completion_cost":26.189401408613186,"max_cost":26.189401408613186},"per_request_limits":null,"top_provider":{"context_length":131000,"max_completion_tokens":131000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"ibm-granite/granite-4.0-h-micro","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","created":1760624583,"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":0.0,"max_prompt_cost":1.1605,"max_completion_cost":0.29708799999999996,"max_cost":1.086228},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0004229838666262653,"input_cache_write":0.0,"max_prompt_cost":1784.9919171628396,"max_completion_cost":456.95793079368684,"max_cost":1670.7524344644175},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-image-mini","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","created":1760463746,"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.0889000000000004e-07,"completion":2.4370500000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027379630080000005,"max_completion_cost":0.07985725440000001,"max_cost":0.10039197696000002},"sats_pricing":{"prompt":0.00032129854508931117,"completion":0.0037484830260419632,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.113242901946194,"max_completion_cost":122.83029179734305,"max_cost":154.4152239738027},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-8b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","created":1760463308,"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.357785e-07,"completion":5.280275000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017796759552,"max_completion_cost":0.017302405120000003,"max_cost":0.030649974784000004},"sats_pricing":{"prompt":0.00020884405430805222,"completion":0.0008121713223090921,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.37360788626502,"max_completion_cost":26.61322988942433,"max_cost":47.143435804123094},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-image","name":"OpenAI: GPT-5 Image","created":1760447986,"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.3750000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":4.642,"max_completion_cost":1.48544,"max_cost":4.642},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0021149193331313266,"input_cache_write":0.0,"max_prompt_cost":7139.9676686513585,"max_completion_cost":2284.789653968435,"max_cost":7139.9676686513585},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-image","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-image","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","created":1759870431,"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.3100000000000002e-07,"completion":1.925e-06,"request":0.0,"image":3.3e-07,"web_search":0.015400000000000002,"internal_reasoning":2.7500000000000004e-06,"input_cache_read":3.3e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.0075694080000000006,"max_completion_cost":0.0157696,"max_cost":0.021446656},"sats_pricing":{"prompt":0.00035530644796606285,"completion":0.0029608870663838573,"request":0.001,"image":0.0005075806399515183,"web_search":23.687096531070857,"internal_reasoning":0.004229838666262653,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":11.642681686951947,"max_completion_cost":24.25558684781656,"max_cost":32.987598113030515},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-flash-image","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","created":1759794479,"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":2.7852000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030421811200000003,"max_completion_cost":0.09126543360000001,"max_cost":0.11408179200000002},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.004283980601190816,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.79249211327354,"max_completion_cost":140.37747633982065,"max_cost":175.47184542477578},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.011408179200000002,"max_cost":0.054188851200000006},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":17.54718454247758,"max_cost":83.3491265767685},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.74075e-05,"completion":0.00013926,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.962999999999999,"max_completion_cost":17.82528,"max_cost":22.560119999999998},"sats_pricing":{"prompt":0.02677487875744259,"completion":0.21419903005954072,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10709.951502977035,"max_completion_cost":27417.475847621212,"max_cost":34700.2428696456},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.8025e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.117646848,"max_completion_cost":0.304218112,"max_cost":0.345810432},"sats_pricing":{"prompt":0.0008924959585814196,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":180.9553405943,"max_completion_cost":467.92492113273534,"max_cost":531.8990314438515},"per_request_limits":null,"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.6","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-07,"input_cache_write":4.125e-06,"max_prompt_cost":3.4815000000000005,"max_completion_cost":1.11408,"max_cost":4.372764},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0005075806399515183,"input_cache_write":0.006344757999393979,"max_prompt_cost":5354.975751488519,"max_completion_cost":1713.5922404763257,"max_cost":6725.849543869579},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","created":1759150481,"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":3.1333500000000005e-07,"completion":4.7580500000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051336806400000004,"max_completion_cost":0.031182356480000003,"max_cost":0.06198444032000001},"sats_pricing":{"prompt":0.0004819478176339667,"completion":0.0007318466860367643,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":78.9623304411491,"max_completion_cost":47.96230441610538,"max_cost":95.33970268079484},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v3.2-exp","alias_ids":null,"forwarded_model_id":null},{"id":"cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","created":1758931878,"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":5.8025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.65e-07,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.076054528,"max_cost":0.076054528},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0008924959585814196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00025379031997575915,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":116.98123028318383,"max_cost":116.98123028318383},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/cydonia-24b-v4.1","alias_ids":null,"forwarded_model_id":null},{"id":"relace-apply-3","name":"Relace: Relace Apply 3","created":1758891572,"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.86425e-07,"completion":1.4506250000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2525248,"max_completion_cost":0.18568,"max_cost":0.3119424},"sats_pricing":{"prompt":0.0015172431295884135,"completion":0.0022312398964535497,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":388.41424117463384,"max_completion_cost":285.59870674605435,"max_cost":479.80582733337127},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"relace/relace-apply-3","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","created":1758668690,"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":4.642e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.060843622400000005,"max_completion_cost":0.152109056,"max_cost":0.1977417728},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.007139967668651357,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":93.58498422654708,"max_completion_cost":233.96246056636767,"max_cost":304.151198736278},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.43705e-07,"completion":2.2049500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.03194290176,"max_completion_cost":0.07225180160000001,"max_cost":0.09620897792000002},"sats_pricing":{"prompt":0.0003748483026041963,"completion":0.0033914846426093955,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":49.13211671893722,"max_completion_cost":111.13216876902467,"max_cost":147.9812563082276},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-max","name":"Qwen: Qwen3 Max","created":1758662808,"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":9.051900000000001e-07,"completion":4.525949999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.716e-07,"input_cache_write":1.0725000000000001e-06,"max_prompt_cost":0.23729012736000002,"max_completion_cost":0.29661265919999996,"max_cost":0.47458025472},"sats_pricing":{"prompt":0.0013922936953870149,"completion":0.006961468476935073,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00026394193277478955,"input_cache_write":0.0016496370798424348,"max_prompt_cost":364.9814384835336,"max_completion_cost":456.22679810441696,"max_cost":729.9628769670671},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-max","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-plus","name":"Qwen: Qwen3 Coder Plus","created":1758662707,"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":7.54325e-07,"completion":3.7716250000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4300000000000002e-07,"input_cache_write":8.9375e-07,"max_prompt_cost":0.754325,"max_completion_cost":0.24717721600000003,"max_cost":0.9520667728000001},"sats_pricing":{"prompt":0.0011602447461558456,"completion":0.005801223730779229,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021995161064565798,"input_cache_write":0.001374697566535362,"max_prompt_cost":1160.2447461558456,"max_completion_cost":380.18899842034756,"max_cost":1464.3959448921237},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-plus","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","created":1758548275,"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":3.1333500000000005e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.485e-07,"input_cache_write":0.0,"max_prompt_cost":0.041069445120000006,"max_completion_cost":0.038027264,"max_cost":0.06882934784},"sats_pricing":{"prompt":0.0004819478176339667,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00022841128797818327,"input_cache_write":0.0,"max_prompt_cost":63.169864352919284,"max_completion_cost":58.49061514159192,"max_cost":105.86801340628138},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-v3.1-terminus","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-flash","name":"Qwen: Qwen3 Coder Flash","created":1758115536,"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.2629750000000002e-07,"completion":1.1314874999999999e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.29e-08,"input_cache_write":2.6812500000000003e-07,"max_prompt_cost":0.2262975,"max_completion_cost":0.07415316479999999,"max_cost":0.28562003184},"sats_pricing":{"prompt":0.0003480734238467537,"completion":0.0017403671192337683,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.598548319369739e-05,"input_cache_write":0.0004124092699606087,"max_prompt_cost":348.0734238467537,"max_completion_cost":114.05669952610424,"max_cost":439.3187834676371},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-flash","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","created":1757612284,"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.36506173440000006,"max_cost":0.36506173440000006},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":561.5099053592826,"max_cost":561.5099053592826},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":262144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.0444500000000002e-07,"completion":1.2765500000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027379630080000005,"max_completion_cost":0.020914995200000005,"max_cost":0.04658339840000001},"sats_pricing":{"prompt":0.00016064927254465559,"completion":0.001963491108879124,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.113242901946194,"max_completion_cost":32.16983832787557,"max_cost":71.65100354845012},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.0173000000000004e-07,"completion":9.051900000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.30173000000000005,"max_completion_cost":0.029661265920000002,"max_cost":0.3215041772800001},"sats_pricing":{"prompt":0.0004640978984623383,"completion":0.0013922936953870149,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":464.0978984623383,"max_completion_cost":45.6226798104417,"max_cost":494.5130183359662},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus-2025-07-28:thinking","name":"Qwen: Qwen Plus 0728 (thinking)","created":1757347599,"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":5.5e-07,"max_prompt_cost":0.46420000000000006,"max_completion_cost":0.04563271680000001,"max_cost":0.4946218112000001},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0008459677332525306,"max_prompt_cost":713.9967668651359,"max_completion_cost":70.18873816991032,"max_cost":760.7892589784094},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-plus-2025-07-28","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-nano-9b-v2","name":"Nemotron Nano 9B V2","created":1757106807,"description":"PPQ.AI model","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.044563200000000004,"max_completion_cost":0.14854399999999998,"max_cost":0.14854399999999998},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":68.54368961905304,"max_completion_cost":228.47896539684342,"max_cost":228.47896539684342},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","created":1757021147,"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.18253086720000003,"max_completion_cost":0.29114624000000006,"max_cost":0.40380200960000007},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":280.7549526796413,"max_completion_cost":447.81877217781323,"max_cost":621.0972195347794},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2-0905","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","created":1756399192,"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":2.7852000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019013632000000003,"max_completion_cost":0.09126543360000001,"max_cost":0.10267361280000001},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.004283980601190816,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":29.245307570795966,"max_completion_cost":140.37747633982065,"max_cost":157.9246608822982},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-70b","name":"Nous: Hermes 4 70B","created":1756236182,"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":null},"pricing":{"prompt":1.5086500000000002e-07,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019774177280000003,"max_completion_cost":0.060843622400000005,"max_cost":0.060843622400000005},"sats_pricing":{"prompt":0.00023204894923116914,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":30.415119873627802,"max_completion_cost":93.58498422654708,"max_cost":93.58498422654708},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-4-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-4-405b","name":"Nous: Hermes 4 405B","created":1756235463,"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":3.4815000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.152109056,"max_completion_cost":0.45632716800000006,"max_cost":0.45632716800000006},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.005354975751488519,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":233.96246056636767,"max_completion_cost":701.8873816991031,"max_cost":701.8873816991031},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-4-405b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":2.90125e-07,"completion":1.1024750000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.4300000000000002e-07,"input_cache_write":0.0,"max_prompt_cost":0.04753408,"max_completion_cost":0.036125900800000006,"max_cost":0.0741531648},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0016957423213046978,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00021995161064565798,"input_cache_write":0.0,"max_prompt_cost":73.1132689269899,"max_completion_cost":55.56608438451234,"max_cost":114.05669952610425},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-chat-v3.1","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","created":1755095639,"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4000000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.060843622400000005,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.767741866020245e-05,"input_cache_write":0.0,"max_prompt_cost":93.58498422654708,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-medium-3.1","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5v","name":"Z.ai: GLM 4.5V","created":1754922288,"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":2.0889e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.21e-07,"input_cache_write":0.0,"max_prompt_cost":0.04563271680000001,"max_completion_cost":0.0342245376,"max_cost":0.06844907520000001},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.003212985450893111,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00018611290131555675,"input_cache_write":0.0,"max_prompt_cost":70.18873816991032,"max_completion_cost":52.64155362743273,"max_cost":105.28310725486548},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.5v","alias_ids":null,"forwarded_model_id":null},{"id":"jamba-large-1.7","name":"AI21: Jamba Large 1.7","created":1754669020,"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":9.284e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5941759999999999,"max_completion_cost":0.038027264,"max_cost":0.622696448},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.014279935337302714,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":913.9158615873737,"max_completion_cost":58.49061514159192,"max_cost":957.7838229435678},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"ai21/jamba-large-1.7","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5","name":"OpenAI: GPT-5","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.4506250000000002e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.375e-07,"input_cache_write":0.0,"max_prompt_cost":0.58025,"max_completion_cost":1.48544,"max_cost":1.8800100000000002},"sats_pricing":{"prompt":0.0022312398964535497,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00021149193331313266,"input_cache_write":0.0,"max_prompt_cost":892.4959585814198,"max_completion_cost":2284.789653968435,"max_cost":2891.6869058038},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini","name":"OpenAI: GPT-5 Mini","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.11605,"max_completion_cost":0.29708799999999996,"max_cost":0.37600199999999995},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.0,"max_prompt_cost":178.49919171628395,"max_completion_cost":456.95793079368684,"max_cost":578.3373811607598},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano","name":"OpenAI: GPT-5 Nano","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.8025000000000005e-08,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5000000000000004e-09,"input_cache_write":0.0,"max_prompt_cost":0.02321,"max_completion_cost":0.05941760000000001,"max_cost":0.07520040000000001},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":8.459677332525306e-06,"input_cache_write":0.0,"max_prompt_cost":35.69983834325679,"max_completion_cost":91.3915861587374,"max_cost":115.66747623215201},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-120b","name":"GPT-OSS 120B (Private via TEE)","created":1786172188,"description":"PPQ.AI model","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.74075e-07,"completion":6.963e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0228163584,"max_completion_cost":0.0912654336,"max_cost":0.0912654336},"sats_pricing":{"prompt":0.0002677487875744259,"completion":0.0010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":35.094369084955154,"max_completion_cost":140.37747633982062,"max_cost":140.37747633982062},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.4815e-08,"completion":1.5086500000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.00456327168,"max_completion_cost":0.019774177280000003,"max_cost":0.019774177280000003},"sats_pricing":{"prompt":5.354975751488518e-05,"completion":0.00023204894923116914,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":7.0188738169910305,"max_completion_cost":30.415119873627802,"max_cost":30.415119873627802},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-oss-20b","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1","name":"Anthropic: Claude Opus 4.1","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.74075e-05,"completion":8.703750000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.65e-06,"input_cache_write":2.0625e-05,"max_prompt_cost":3.4814999999999996,"max_completion_cost":2.7852,"max_cost":5.7096599999999995},"sats_pricing":{"prompt":0.02677487875744259,"completion":0.13387439378721297,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0025379031997575918,"input_cache_write":0.0317237899969699,"max_prompt_cost":5354.975751488518,"max_completion_cost":4283.980601190815,"max_cost":8782.160232441169},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"codestral-2508","name":"Mistral: Codestral 2508","created":1754079630,"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.04445e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":0.0,"max_prompt_cost":0.08912640000000001,"max_completion_cost":0.2673792,"max_cost":0.2673792},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0016064927254465554,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0,"max_prompt_cost":137.08737923810608,"max_completion_cost":411.26213771431816,"max_cost":411.26213771431816},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/codestral-2508","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","created":1753972379,"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":8.123500000000001e-08,"completion":3.1333500000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.012997600000000002,"max_completion_cost":0.010267361280000002,"max_cost":0.020603052800000004},"sats_pricing":{"prompt":0.00012494943420139878,"completion":0.0004819478176339667,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":19.991909472223803,"max_completion_cost":15.792466088229821,"max_cost":31.69003250054219},"per_request_limits":null,"top_provider":{"context_length":160000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","created":1753806965,"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":5.5878075000000007e-08,"completion":2.2403452500000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007152393600000001,"max_completion_cost":0.007169104800000001,"max_cost":0.012533400000000002},"sats_pricing":{"prompt":8.594736081139073e-05,"completion":0.0003445926896082862,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.001262183858014,"max_completion_cost":11.026966067465159,"max_cost":19.27791270535867},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5","name":"Z.ai: GLM 4.5","created":1753471347,"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963000000000001e-07,"completion":2.5531000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.21e-07,"input_cache_write":0.0,"max_prompt_cost":0.09126543360000001,"max_completion_cost":0.25097994240000004,"max_cost":0.27379630080000006},"sats_pricing":{"prompt":0.001070995150297704,"completion":0.003926982217758248,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00018611290131555675,"input_cache_write":0.0,"max_prompt_cost":140.37747633982065,"max_completion_cost":386.03805993450675,"max_cost":421.13242901946194},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.5","alias_ids":null,"forwarded_model_id":null},{"id":"glm-4.5-air","name":"Z.ai: GLM 4.5 Air","created":1753471258,"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.5086500000000002e-07,"completion":9.86425e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.019774177280000003,"max_completion_cost":0.0969695232,"max_cost":0.10191306752000001},"sats_pricing":{"prompt":0.00023204894923116914,"completion":0.0015172431295884135,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.0,"max_prompt_cost":30.415119873627802,"max_completion_cost":149.1510686110594,"max_cost":156.75484857946637},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"z-ai/glm-4.5-air","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","created":1753449557,"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.6691500000000003e-07,"completion":2.66915e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.034985082880000004,"max_completion_cost":0.3498508288,"max_cost":0.3498508288},"sats_pricing":{"prompt":0.0004105481409474531,"completion":0.004105481409474531,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":53.81136593026457,"max_completion_cost":538.1136593026457,"max_cost":538.1136593026457},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":3.4815000000000006e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.09126543360000001,"max_completion_cost":0.076054528,"max_cost":0.1445036032},"sats_pricing":{"prompt":0.000535497575148852,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":140.37747633982065,"max_completion_cost":116.98123028318383,"max_cost":222.2643375380493},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","created":1753205056,"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":2.3210000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.014854400000000002,"max_completion_cost":0.00047534080000000004,"max_cost":0.015092070400000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0003569983834325679,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":22.84789653968435,"max_completion_cost":0.731132689269899,"max_cost":23.213462884319295},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"bytedance/ui-tars-1.5-7b","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":7.700000000000001e-08,"completion":3.0800000000000006e-07,"request":0.0,"image":1.1e-07,"web_search":0.015400000000000002,"internal_reasoning":4.4e-07,"input_cache_read":1.1000000000000001e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.08074035200000002,"max_completion_cost":0.020184780000000003,"max_cost":0.09587893700000001},"sats_pricing":{"prompt":0.0001184354826553543,"completion":0.0004737419306214172,"request":0.001,"image":0.0001691935466505061,"web_search":23.687096531070857,"internal_reasoning":0.0006767741866020244,"input_cache_read":1.6919354665050612e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":124.18860466082079,"max_completion_cost":31.046677423274573,"max_cost":147.47361272827672},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":1.0444500000000002e-07,"completion":6.382750000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027379630080000005,"max_completion_cost":0.010457497600000002,"max_cost":0.036125900800000006},"sats_pricing":{"prompt":0.00016064927254465559,"completion":0.000981745554439562,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.113242901946194,"max_completion_cost":16.084919163937784,"max_cost":55.56608438451234},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-235b-a22b-07-25","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2","name":"MoonshotAI: Kimi K2 0711","created":1752263252,"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.614850000000002e-07,"completion":2.66915e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.08670216192000002,"max_completion_cost":0.2678545408,"max_cost":0.28817536},"sats_pricing":{"prompt":0.0010174453927828187,"completion":0.004105481409474531,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":133.3586025228296,"max_completion_cost":411.9932704035881,"max_cost":443.24919286987625},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":100352,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"moonshotai/kimi-k2","alias_ids":null,"forwarded_model_id":null},{"id":"dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","created":1752094966,"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":1.04445e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.029708800000000004,"max_completion_cost":0.0085561344,"max_cost":0.0363635712},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.0016064927254465554,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":45.6957930793687,"max_completion_cost":13.160388406858182,"max_cost":55.93165072914727},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"venice/uncensored","alias_ids":null,"forwarded_model_id":null},{"id":"hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","created":1751987664,"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.6247000000000002e-07,"completion":6.614850000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.021295267840000003,"max_completion_cost":0.08670216192000002,"max_cost":0.08670216192000002},"sats_pricing":{"prompt":0.00024989886840279756,"completion":0.0010174453927828187,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.75474447929148,"max_completion_cost":133.3586025228296,"max_cost":133.3586025228296},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"tencent/hunyuan-a13b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-large","name":"Morph: Morph V3 Large","created":1751910858,"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.04445e-06,"completion":2.2049500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.2737963008,"max_completion_cost":0.28900720640000005,"max_cost":0.4259053568000001},"sats_pricing":{"prompt":0.0016064927254465554,"completion":0.0033914846426093955,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":421.1324290194618,"max_completion_cost":444.5286750760987,"max_cost":655.0948895858296},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"morph/morph-v3-large","alias_ids":null,"forwarded_model_id":null},{"id":"morph-v3-fast","name":"Morph: Morph V3 Fast","created":1751910002,"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.284000000000001e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.07605452800000001,"max_completion_cost":0.05291880000000001,"max_cost":0.09369412800000002},"sats_pricing":{"prompt":0.0014279935337302716,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":116.98123028318386,"max_completion_cost":81.39563142262548,"max_cost":144.11310742405902},"per_request_limits":null,"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"morph/morph-v3-fast","alias_ids":null,"forwarded_model_id":null},{"id":"ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","created":1751300903,"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":123000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.8741e-07,"completion":1.4506250000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05995143000000001,"max_completion_cost":0.02321,"max_cost":0.07536287},"sats_pricing":{"prompt":0.0007496966052083926,"completion":0.0022312398964535497,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":92.21268244063229,"max_completion_cost":35.69983834325679,"max_cost":115.91737510055479},"per_request_limits":null,"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.0879687500000001e-07,"completion":2.90125e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.027852000000000005,"max_completion_cost":0.004753408,"max_cost":0.030822880000000004},"sats_pricing":{"prompt":0.0001673429922340162,"completion":0.0004462479792907098,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":42.83980601190815,"max_completion_cost":7.31132689269899,"max_cost":47.40938531984502},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m1","name":"MiniMax: MiniMax M1","created":1750200414,"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.382750000000001e-07,"completion":2.5531000000000006e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6382750000000001,"max_completion_cost":0.10212400000000002,"max_cost":0.7148680000000002},"sats_pricing":{"prompt":0.000981745554439562,"completion":0.003926982217758248,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":981.7455544395619,"max_completion_cost":157.0792887103299,"max_cost":1099.5550209723094},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-m1","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":2.3100000000000002e-07,"completion":1.925e-06,"request":0.0,"image":3.3e-07,"web_search":0.015400000000000002,"internal_reasoning":2.7500000000000004e-06,"input_cache_read":3.3e-08,"input_cache_write":9.166666666666664e-08,"max_prompt_cost":0.24222105600000002,"max_completion_cost":0.126154875,"max_cost":0.353237346},"sats_pricing":{"prompt":0.00035530644796606285,"completion":0.0029608870663838573,"request":0.001,"image":0.0005075806399515183,"web_search":23.687096531070857,"internal_reasoning":0.004229838666262653,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.00014099462220875506,"max_prompt_cost":372.5658139824623,"max_completion_cost":194.04173389546605,"max_cost":543.3225398104724},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.625e-07,"completion":7.7e-06,"request":0.0,"image":1.3750000000000002e-06,"web_search":0.015400000000000002,"internal_reasoning":1.1000000000000001e-05,"input_cache_read":1.375e-07,"input_cache_write":4.125e-07,"max_prompt_cost":1.0092544,"max_completion_cost":0.5046272,"max_cost":1.4508032000000002},"sats_pricing":{"prompt":0.0014804435331919287,"completion":0.01184354826553543,"request":0.001,"image":0.0021149193331313266,"web_search":23.687096531070857,"internal_reasoning":0.016919354665050613,"input_cache_read":0.00021149193331313266,"input_cache_write":0.0006344757999393979,"max_prompt_cost":1552.3575582602598,"max_completion_cost":776.1787791301299,"max_cost":2231.513989999123},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro","name":"OpenAI: o3 Pro","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.3210000000000003e-05,"completion":9.284000000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.642,"max_completion_cost":9.284,"max_cost":11.605},"sats_pricing":{"prompt":0.035699838343256796,"completion":0.14279935337302718,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7139.9676686513585,"max_completion_cost":14279.935337302717,"max_cost":17849.919171628393},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","created":1749137257,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.625e-07,"completion":7.7e-06,"request":0.0,"image":1.3750000000000002e-06,"web_search":0.015400000000000002,"internal_reasoning":1.1000000000000001e-05,"input_cache_read":1.375e-07,"input_cache_write":4.125e-07,"max_prompt_cost":1.0092544,"max_completion_cost":0.5046272,"max_cost":1.4508032000000002},"sats_pricing":{"prompt":0.0014804435331919287,"completion":0.01184354826553543,"request":0.001,"image":0.0021149193331313266,"web_search":23.687096531070857,"internal_reasoning":0.016919354665050613,"input_cache_read":0.00021149193331313266,"input_cache_write":0.0006344757999393979,"max_prompt_cost":1552.3575582602598,"max_completion_cost":776.1787791301299,"max_cost":2231.513989999123},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-pro-preview-06-05","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":5.8025e-07,"completion":2.4950750000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.85e-07,"input_cache_write":0.0,"max_prompt_cost":0.09506816,"max_completion_cost":0.08175861760000001,"max_cost":0.1578131456},"sats_pricing":{"prompt":0.0008924959585814196,"completion":0.0038377326219001056,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0005921774132767714,"input_cache_write":0.0,"max_prompt_cost":146.2265378539798,"max_completion_cost":125.75482255442266,"max_cost":242.73605283760648},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-r1-0528","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","created":1747931245,"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.74075e-05,"completion":8.703750000000001e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.65e-06,"input_cache_write":2.0625e-05,"max_prompt_cost":3.4814999999999996,"max_completion_cost":2.7852,"max_cost":5.7096599999999995},"sats_pricing":{"prompt":0.02677487875744259,"completion":0.13387439378721297,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0025379031997575918,"input_cache_write":0.0317237899969699,"max_prompt_cost":5354.975751488518,"max_completion_cost":4283.980601190815,"max_cost":8782.160232441169},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4-opus-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","created":1747930371,"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-07,"input_cache_write":4.125e-06,"max_prompt_cost":0.6963000000000001,"max_completion_cost":1.11408,"max_cost":1.587564},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0005075806399515183,"input_cache_write":0.006344757999393979,"max_prompt_cost":1070.995150297704,"max_completion_cost":1713.5922404763257,"max_cost":2441.868942678764},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-4-sonnet-20250522","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3n-e4b-it","name":"Google: Gemma 3n 4B","created":1747776824,"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.963e-08,"completion":1.3926e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00228163584,"max_completion_cost":0.00456327168,"max_cost":0.00456327168},"sats_pricing":{"prompt":0.00010709951502977036,"completion":0.00021419903005954073,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.5094369084955153,"max_completion_cost":7.0188738169910305,"max_cost":7.0188738169910305},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3n-e4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-medium-3","name":"Mistral: Mistral Medium 3","created":1746627341,"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4000000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.060843622400000005,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.767741866020245e-05,"input_cache_write":0.0,"max_prompt_cost":93.58498422654708,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-medium-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","created":1746578513,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.625e-07,"completion":7.7e-06,"request":0.0,"image":1.3750000000000002e-06,"web_search":0.015400000000000002,"internal_reasoning":1.1000000000000001e-05,"input_cache_read":1.375e-07,"input_cache_write":4.125e-07,"max_prompt_cost":1.0092544,"max_completion_cost":0.5046195,"max_cost":1.4507964625},"sats_pricing":{"prompt":0.0014804435331919287,"completion":0.01184354826553543,"request":0.001,"image":0.0021149193331313266,"web_search":23.687096531070857,"internal_reasoning":0.016919354665050613,"input_cache_read":0.00021149193331313266,"input_cache_write":0.0006344757999393979,"max_prompt_cost":1552.3575582602598,"max_completion_cost":776.1669355818642,"max_cost":2231.503626894391},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemini-2.5-pro-preview-03-25","alias_ids":null,"forwarded_model_id":null},{"id":"virtuoso-large","name":"Arcee AI: Virtuoso Large","created":1746478885,"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.703750000000001e-07,"completion":1.3926000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.11408179200000002,"max_completion_cost":0.08912640000000001,"max_cost":0.147504192},"sats_pricing":{"prompt":0.0013387439378721297,"completion":0.002141990300595408,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":175.47184542477578,"max_completion_cost":137.08737923810608,"max_cost":226.87961263906556},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":64000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"arcee-ai/virtuoso-large","alias_ids":null,"forwarded_model_id":null},{"id":"llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.0889000000000004e-07,"completion":2.0889000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.03422453760000001,"max_completion_cost":0.0034224537600000006,"max_cost":0.03422453760000001},"sats_pricing":{"prompt":0.00032129854508931117,"completion":0.00032129854508931117,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":52.64155362743274,"max_completion_cost":5.264155362743274,"max_cost":52.64155362743274},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-guard-4-12b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-30b-a3b","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.3926e-07,"completion":5.8025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0057040896,"max_completion_cost":0.009506816,"max_cost":0.012929269759999999},"sats_pricing":{"prompt":0.00021419903005954073,"completion":0.0008924959585814196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.773592271238789,"max_completion_cost":14.62265378539798,"max_cost":19.886809148141253},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-30b-a3b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-8b","name":"Qwen: Qwen3 8B","created":1745876632,"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":1.357785e-07,"completion":5.280275000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.017796759552,"max_completion_cost":0.004325601280000001,"max_cost":0.02101006336},"sats_pricing":{"prompt":0.00020884405430805222,"completion":0.0008121713223090921,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":27.37360788626502,"max_completion_cost":6.653307472356082,"max_cost":32.31606486572954},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-8b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-14b","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":2.6401375000000004e-07,"completion":1.0560550000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.034604810240000006,"max_completion_cost":0.008651202560000001,"max_cost":0.04109321216},"sats_pricing":{"prompt":0.00040608566115454603,"completion":0.0016243426446181841,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":53.22645977884866,"max_completion_cost":13.306614944712164,"max_cost":63.20642098738278},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-14b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-32b","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":9.284e-08,"completion":3.2494000000000005e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0038027264000000003,"max_completion_cost":0.005323816960000001,"max_cost":0.007605452800000001},"sats_pricing":{"prompt":0.00014279935337302717,"completion":0.0004997977368055951,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.849061514159192,"max_completion_cost":8.18868611982287,"max_cost":11.698123028318385},"per_request_limits":null,"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-32b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-235b-a22b","name":"Qwen: Qwen3 235B A22B","created":1745875757,"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":5.280275000000001e-07,"completion":2.1121100000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.06920962048000001,"max_completion_cost":0.017302405120000003,"max_cost":0.08218642432},"sats_pricing":{"prompt":0.0008121713223090921,"completion":0.0032486852892363682,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":106.45291955769731,"max_completion_cost":26.61322988942433,"max_cost":126.41284197476556},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen3-235b-a22b-04-28","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high","name":"OpenAI: o4 Mini High","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.2765500000000003e-06,"completion":5.106200000000001e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.025e-07,"input_cache_write":0.0,"max_prompt_cost":0.25531000000000004,"max_completion_cost":0.5106200000000001,"max_cost":0.6382750000000001},"sats_pricing":{"prompt":0.001963491108879124,"completion":0.007853964435516496,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00046528225328889184,"input_cache_write":0.0,"max_prompt_cost":392.6982217758247,"max_completion_cost":785.3964435516494,"max_cost":981.7455544395619},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3","name":"OpenAI: o3","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":9.284e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":0.0,"max_prompt_cost":0.4642,"max_completion_cost":0.9284,"max_cost":1.1605},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.014279935337302714,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.0,"max_prompt_cost":713.9967668651358,"max_completion_cost":1427.9935337302716,"max_cost":1784.9919171628396},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini","name":"OpenAI: o4 Mini","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.2765500000000003e-06,"completion":5.106200000000001e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.025e-07,"input_cache_write":0.0,"max_prompt_cost":0.25531000000000004,"max_completion_cost":0.5106200000000001,"max_cost":0.6382750000000001},"sats_pricing":{"prompt":0.001963491108879124,"completion":0.007853964435516496,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.00046528225328889184,"input_cache_write":0.0,"max_prompt_cost":392.6982217758247,"max_completion_cost":785.3964435516494,"max_cost":981.7455544395619},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1","name":"OpenAI: GPT-4.1","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":9.284e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":5.5e-07,"input_cache_write":0.0,"max_prompt_cost":2.431423896,"max_completion_cost":0.304218112,"max_cost":2.65958748},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.014279935337302714,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0008459677332525306,"input_cache_write":0.0,"max_prompt_cost":3739.8293852275574,"max_completion_cost":467.92492113273534,"max_cost":4090.7730760771087},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini","name":"OpenAI: GPT-4.1 Mini","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":1.8568000000000002e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":1.1e-07,"input_cache_write":0.0,"max_prompt_cost":0.4862847792,"max_completion_cost":0.060843622400000005,"max_cost":0.531917496},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.002855987067460543,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0001691935466505061,"input_cache_write":0.0,"max_prompt_cost":747.9658770455115,"max_completion_cost":93.58498422654708,"max_cost":818.1546152154218},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano","name":"OpenAI: GPT-4.1 Nano","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.1215711948,"max_completion_cost":0.015210905600000001,"max_cost":0.132979374},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.0,"max_prompt_cost":186.9914692613779,"max_completion_cost":23.39624605663677,"max_cost":204.53865380385545},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-maverick","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":9.284000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.24337448960000002,"max_completion_cost":0.015210905600000001,"max_cost":0.2547826688},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.0014279935337302716,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":374.3399369061883,"max_completion_cost":23.39624605663677,"max_cost":391.8871214486659},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-4-scout","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":1310720,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":3.4815000000000006e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.038027264000000005,"max_completion_cost":0.005704089600000001,"max_cost":0.04182999040000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.000535497575148852,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":58.49061514159193,"max_completion_cost":8.77359227123879,"max_cost":64.33967665575113},"per_request_limits":null,"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","created":1742824755,"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":3.1333500000000005e-07,"completion":1.2997600000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.485e-07,"input_cache_write":0.0,"max_prompt_cost":0.051336806400000004,"max_completion_cost":0.08518107136000001,"max_cost":0.11598315520000002},"sats_pricing":{"prompt":0.0004819478176339667,"completion":0.0019991909472223805,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00022841128797818327,"input_cache_write":0.0,"max_prompt_cost":78.9623304411491,"max_completion_cost":131.01897791716593,"max_cost":178.3963761818554},"per_request_limits":null,"top_provider":{"context_length":163840,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-chat-v3-0324","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro","name":"OpenAI: o1-pro","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":0.00017407500000000002,"completion":0.0006963000000000001,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.815000000000005,"max_completion_cost":69.63000000000001,"max_cost":87.03750000000001},"sats_pricing":{"prompt":0.26774878757442594,"completion":1.0709951502977038,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":53549.75751488519,"max_completion_cost":107099.51502977038,"max_cost":133874.39378721296},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","created":1742238937,"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":4.0733550000000003e-07,"completion":6.440775e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.052138944000000007,"max_completion_cost":0.08244192,"max_cost":0.08244192},"sats_pricing":{"prompt":0.0006265321629241567,"completion":0.000990670514025376,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":80.19611685429206,"max_completion_cost":126.80582579524811,"max_cost":126.80582579524811},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":128000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.8025000000000005e-08,"completion":1.1605000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.0019013632000000002,"max_cost":0.0085561344},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.00017849919171628395,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":2.924530757079596,"max_cost":13.160388406858182},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3-4b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":5.8025000000000005e-08,"completion":1.7407500000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.0028520448000000005,"max_cost":0.009506816000000001},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.000267748787574426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":4.386796135619395,"max_cost":14.622653785397983},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3-12b-it","alias_ids":null,"forwarded_model_id":null},{"id":"command-a","name":"Cohere: Command A","created":1741894342,"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.74272,"max_completion_cost":0.09506816000000001,"max_cost":0.81402112},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1142.3948269842174,"max_completion_cost":146.22653785397983,"max_cost":1252.064730374702},"per_request_limits":null,"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-a-03-2025","alias_ids":null,"forwarded_model_id":null},{"id":"reka-flash-3","name":"Reka Flash 3","created":1741812813,"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605000000000001e-07,"completion":2.3210000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.015210905600000001,"max_cost":0.015210905600000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0003569983834325679,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":23.39624605663677,"max_cost":23.39624605663677},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"rekaai/reka-flash-3","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":9.284e-08,"completion":5.22225e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.4000000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.01216872448,"max_completion_cost":0.0684490752,"max_cost":0.0684490752},"sats_pricing":{"prompt":0.00014279935337302717,"completion":0.0008032463627232777,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.767741866020245e-05,"input_cache_write":0.0,"max_prompt_cost":18.716996845309417,"max_completion_cost":105.28310725486546,"max_cost":105.28310725486546},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-3-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","created":1741636566,"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":6.382750000000001e-07,"completion":9.284000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-07,"input_cache_write":0.0,"max_prompt_cost":0.020914995200000005,"max_completion_cost":0.030421811200000003,"max_cost":0.030421811200000003},"sats_pricing":{"prompt":0.000981745554439562,"completion":0.0014279935337302716,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0004229838666262653,"input_cache_write":0.0,"max_prompt_cost":32.16983832787557,"max_completion_cost":46.79249211327354,"max_cost":46.79249211327354},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/skyfall-36b-v2","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","created":1741313308,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.321e-06,"completion":9.284e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.29708799999999996,"max_completion_cost":1.1883519999999999,"max_cost":1.1883519999999999},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.014279935337302714,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":456.95793079368684,"max_completion_cost":1827.8317231747474,"max_cost":1827.8317231747474},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-reasoning-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-pro","name":"Perplexity: Sonar Pro","created":1741312423,"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.6963000000000001,"max_completion_cost":0.13926,"max_cost":0.8077080000000001},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1070.995150297704,"max_completion_cost":214.19903005954072,"max_cost":1242.3543743453363},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-pro","alias_ids":null,"forwarded_model_id":null},{"id":"sonar-deep-research","name":"Perplexity: Sonar Deep Research","created":1741311246,"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":2.321e-06,"completion":9.284e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":3.3e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.29708799999999996,"max_completion_cost":1.1883519999999999,"max_cost":1.1883519999999999},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.014279935337302714,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0050758063995151835,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":456.95793079368684,"max_completion_cost":1827.8317231747474,"max_cost":1827.8317231747474},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar-deep-research","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-saba","name":"Mistral: Saba","created":1739803239,"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2000000000000002e-08,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.022816358400000004,"max_cost":0.022816358400000004},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.3838709330101225e-05,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":35.09436908495516,"max_cost":35.09436908495516},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-saba-2502","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high","name":"OpenAI: o3 Mini High","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.2765500000000003e-06,"completion":5.106200000000001e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":6.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.25531000000000004,"max_completion_cost":0.5106200000000001,"max_cost":0.6382750000000001},"sats_pricing":{"prompt":0.001963491108879124,"completion":0.007853964435516496,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0009305645065777837,"input_cache_write":0.0,"max_prompt_cost":392.6982217758247,"max_completion_cost":785.3964435516494,"max_cost":981.7455544395619},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","created":1738696718,"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.284000000000001e-07,"completion":1.8568000000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.030421811200000003,"max_completion_cost":0.060843622400000005,"max_cost":0.060843622400000005},"sats_pricing":{"prompt":0.0014279935337302716,"completion":0.002855987067460543,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":46.79249211327354,"max_completion_cost":93.58498422654708,"max_cost":93.58498422654708},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","created":1738410311,"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":8.703750000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.009283999999999999,"max_completion_cost":0.027852000000000005,"max_cost":0.027852000000000005},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0013387439378721297,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14.279935337302714,"max_completion_cost":42.83980601190815,"max_cost":42.83980601190815},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-plus","name":"Qwen: Qwen-Plus","created":1738409840,"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":3.0173000000000004e-07,"completion":9.051900000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":5.720000000000001e-08,"input_cache_write":3.575e-07,"max_prompt_cost":0.30173000000000005,"max_completion_cost":0.029661265920000002,"max_cost":0.3215041772800001},"sats_pricing":{"prompt":0.0004640978984623383,"completion":0.0013922936953870149,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.798064425826319e-05,"input_cache_write":0.0005498790266141449,"max_prompt_cost":464.0978984623383,"max_completion_cost":45.6226798104417,"max_cost":494.5130183359662},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-plus-2025-01-25","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini","name":"OpenAI: o3 Mini","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.2765500000000003e-06,"completion":5.106200000000001e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":6.05e-07,"input_cache_write":0.0,"max_prompt_cost":0.25531000000000004,"max_completion_cost":0.5106200000000001,"max_cost":0.6382750000000001},"sats_pricing":{"prompt":0.001963491108879124,"completion":0.007853964435516496,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.0009305645065777837,"input_cache_write":0.0,"max_prompt_cost":392.6982217758247,"max_completion_cost":785.3964435516494,"max_cost":981.7455544395619},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":5.8025000000000005e-08,"completion":9.284e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0019013632000000002,"max_completion_cost":0.00152109056,"max_cost":0.00247177216},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.00014279935337302717,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.924530757079596,"max_completion_cost":2.339624605663677,"max_cost":3.801889984203475},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","alias_ids":null,"forwarded_model_id":null},{"id":"sonar","name":"Perplexity: Sonar","created":1738013808,"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0055000000000000005,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14746705599999999,"max_completion_cost":0.14746705599999999,"max_cost":0.14746705599999999},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":8.459677332525306,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":226.8224928977163,"max_completion_cost":226.8224928977163,"max_cost":226.8224928977163},"per_request_limits":null,"top_provider":{"context_length":127072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"perplexity/sonar","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","created":1737663169,"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"deepseek-r1"},"pricing":{"prompt":9.284000000000001e-07,"completion":9.284000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.007605452800000001,"max_cost":0.007605452800000001},"sats_pricing":{"prompt":0.0014279935337302716,"completion":0.0014279935337302716,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":11.698123028318385,"max_cost":11.698123028318385},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":8192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-r1","name":"DeepSeek: R1","created":1737381095,"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":8.1235e-07,"completion":2.9012500000000004e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.051990400000000006,"max_completion_cost":0.04642,"max_cost":0.08541280000000001},"sats_pricing":{"prompt":0.0012494943420139877,"completion":0.0044624797929070995,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":79.96763788889521,"max_completion_cost":71.39967668651359,"max_cost":131.375405103185},"per_request_limits":null,"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-r1","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-01","name":"MiniMax: MiniMax-01","created":1736915462,"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.3210000000000002e-07,"completion":1.2765500000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.23214456320000001,"max_completion_cost":1.2767950976000002,"max_cost":1.2767950976000002},"sats_pricing":{"prompt":0.0003569983834325679,"completion":0.001963491108879124,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":357.06692712218694,"max_completion_cost":1963.8680991720285,"max_cost":1963.8680991720285},"per_request_limits":null,"top_provider":{"context_length":1000192,"max_completion_tokens":1000192,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"minimax/minimax-01","alias_ids":null,"forwarded_model_id":null},{"id":"phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.123500000000001e-08,"completion":1.6247000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0013309542400000002,"max_completion_cost":0.0026619084800000004,"max_cost":0.0026619084800000004},"sats_pricing":{"prompt":0.00012494943420139878,"completion":0.00024989886840279756,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.0471715299557176,"max_completion_cost":4.094343059911435,"max_cost":4.094343059911435},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"microsoft/phi-4","alias_ids":null,"forwarded_model_id":null},{"id":"deepseek-chat","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":2.987127e-07,"completion":1.19380635e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.0800000000000004e-08,"input_cache_write":0.0,"max_prompt_cost":0.0382352256,"max_completion_cost":0.0191009016,"max_cost":0.052556724},"sats_pricing":{"prompt":0.0004594569194777149,"completion":0.001836221185185413,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.737419306214172e-05,"input_cache_write":0.0,"max_prompt_cost":58.8104856931475,"max_completion_cost":29.37953896296661,"max_cost":80.83871394447067},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"deepseek/deepseek-chat-v3","alias_ids":null,"forwarded_model_id":null},{"id":"l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","created":1734535928,"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":7.54325e-07,"completion":8.703750000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0988708864,"max_completion_cost":0.014260224000000002,"max_cost":0.1007722496},"sats_pricing":{"prompt":0.0011602447461558456,"completion":0.0013387439378721297,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":152.075599368139,"max_completion_cost":21.933980678096972,"max_cost":155.0001301252186},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","alias_ids":null,"forwarded_model_id":null},{"id":"o1","name":"OpenAI: o1","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.74075e-05,"completion":6.963e-05,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":8.25e-06,"input_cache_write":0.0,"max_prompt_cost":3.4814999999999996,"max_completion_cost":6.962999999999999,"max_cost":8.70375},"sats_pricing":{"prompt":0.02677487875744259,"completion":0.10709951502977036,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":0.012689515998787959,"input_cache_write":0.0,"max_prompt_cost":5354.975751488518,"max_completion_cost":10709.951502977035,"max_cost":13387.439378721294},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"command-r7b-12-2024","name":"Cohere: Command R7B (12-2024)","created":1734158152,"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":4.351875000000001e-08,"completion":1.7407500000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0055704000000000005,"max_completion_cost":0.0006963000000000001,"max_cost":0.006092625000000002},"sats_pricing":{"prompt":6.69371968936065e-05,"completion":0.000267748787574426,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8.56796120238163,"max_completion_cost":1.0709951502977038,"max_cost":9.371207565104909},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-r7b-12-2024","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.3-70b-instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":1.1605000000000001e-07,"completion":3.7136e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.015210905600000001,"max_completion_cost":0.00608436224,"max_cost":0.01939390464},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0005711974134921087,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":23.39624605663677,"max_completion_cost":9.358498422654709,"max_cost":29.83021372221188},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.3-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"nova-lite-v1","name":"Amazon: Nova Lite 1.0","created":1733437363,"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":6.963e-08,"completion":2.7852e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.020888999999999998,"max_completion_cost":0.0014260224,"max_cost":0.0219585168},"sats_pricing":{"prompt":0.00010709951502977036,"completion":0.00042839806011908145,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":32.1298545089311,"max_completion_cost":2.193398067809697,"max_cost":33.774903059788386},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-lite-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-micro-v1","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":4.0617500000000006e-08,"completion":1.6247000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005199040000000001,"max_completion_cost":0.0008318464000000002,"max_cost":0.0058229248},"sats_pricing":{"prompt":6.247471710069939e-05,"completion":0.00024989886840279756,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.996763788889522,"max_completion_cost":1.2794822062223234,"max_cost":8.956375443556263},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-micro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"nova-pro-v1","name":"Amazon: Nova Pro 1.0","created":1733436303,"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":9.284000000000001e-07,"completion":3.7136000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.27852000000000005,"max_completion_cost":0.019013632000000003,"max_cost":0.29278022400000003},"sats_pricing":{"prompt":0.0014279935337302716,"completion":0.005711974134921086,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":428.39806011908155,"max_completion_cost":29.245307570795966,"max_cost":450.3320407971785},"per_request_limits":null,"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"amazon/nova-pro-v1","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-11-20","name":"OpenAI: GPT-4o (2024-11-20)","created":1732127594,"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3750000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.37136,"max_completion_cost":0.19013632000000003,"max_cost":0.5139622400000001},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0021149193331313266,"input_cache_write":0.0,"max_prompt_cost":571.1974134921087,"max_completion_cost":292.45307570795967,"max_cost":790.5372202730786},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-2024-11-20","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large-2407","name":"Mistral Large 2407","created":1731978415,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.304218112,"max_completion_cost":0.9126543360000001,"max_cost":0.9126543360000001},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":467.92492113273534,"max_completion_cost":1403.7747633982062,"max_cost":1403.7747633982062},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-large-2407","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","created":1731368400,"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":7.6593e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.02509799424,"max_completion_cost":0.038027264,"max_cost":0.038027264},"sats_pricing":{"prompt":0.001178094665327474,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":38.60380599345067,"max_completion_cost":58.49061514159192,"max_cost":58.49061514159192},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","created":1731103448,"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":1024000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":4.6420000000000004e-07,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.47534080000000006,"max_completion_cost":0.47534080000000006,"max_cost":0.47534080000000006},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":731.1326892698992,"max_completion_cost":731.1326892698992,"max_cost":731.1326892698992},"per_request_limits":null,"top_provider":{"context_length":1024000,"max_completion_tokens":1024000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/unslopnemo-12b","alias_ids":null,"forwarded_model_id":null},{"id":"magnum-v4-72b","name":"Magnum v4 72B","created":1729555200,"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":3.4815000000000005e-06,"completion":5.802500000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05704089600000001,"max_completion_cost":0.011883520000000002,"max_cost":0.06179430400000001},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.008924959585814199,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":87.73592271238789,"max_completion_cost":18.27831723174748,"max_cost":95.04724960508689},"per_request_limits":null,"top_provider":{"context_length":16384,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthracite-org/magnum-v4-72b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","created":1729036800,"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":1.1605000000000001e-07,"completion":2.3210000000000002e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0038027264000000003,"max_completion_cost":0.007605452800000001,"max_cost":0.007605452800000001},"sats_pricing":{"prompt":0.00017849919171628395,"completion":0.0003569983834325679,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.849061514159192,"max_completion_cost":11.698123028318385,"max_cost":11.698123028318385},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":32768,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-2.5-7b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"rocinante-12b","name":"TheDrummer: Rocinante 12B","created":1727654400,"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":2.90125e-07,"completion":5.8025e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.019013632,"max_completion_cost":0.038027264,"max_cost":0.038027264},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0008924959585814196,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":29.24530757079596,"max_completion_cost":58.49061514159192,"max_cost":58.49061514159192},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"thedrummer/rocinante-12b","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","created":1727222400,"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":3.13335e-08,"completion":2.3326050000000003e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00188001,"max_completion_cost":0.013995630000000002,"max_cost":0.013995630000000002},"sats_pricing":{"prompt":4.8194781763396666e-05,"completion":0.0003587833753497308,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.8916869058038,"max_completion_cost":21.527002520983846,"max_cost":21.527002520983846},"per_request_limits":null,"top_provider":{"context_length":60000,"max_completion_tokens":60000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.2-1b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","created":1727222400,"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.8025000000000005e-08,"completion":3.82965e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.05019598848,"max_cost":0.05019598848},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.000589047332663737,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":77.20761198690134,"max_cost":77.20761198690134},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.2-3b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":4.177800000000001e-07,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.013689815040000003,"max_completion_cost":0.007605452800000001,"max_cost":0.014450360320000001},"sats_pricing":{"prompt":0.0006425970901786223,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":21.056621450973097,"max_completion_cost":11.698123028318385,"max_cost":22.226433753804933},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"qwen/qwen-2.5-72b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-08-2024","name":"Cohere: Command R (08-2024)","created":1724976000,"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.022281600000000002,"max_completion_cost":0.0027852000000000003,"max_cost":0.024370500000000007},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":34.27184480952652,"max_completion_cost":4.283980601190815,"max_cost":37.484830260419635},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-r-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"command-r-plus-08-2024","name":"Cohere: Command R+ (08-2024)","created":1724976000,"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.37136,"max_completion_cost":0.04642,"max_cost":0.40617500000000006},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":571.1974134921087,"max_completion_cost":71.39967668651359,"max_cost":624.7471710069939},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"cohere/command-r-plus-08-2024","alias_ids":null,"forwarded_model_id":null},{"id":"l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":9.86425e-07,"completion":9.86425e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1292926976,"max_completion_cost":0.0161615872,"max_cost":0.1292926976},"sats_pricing":{"prompt":0.0015172431295884135,"completion":0.0015172431295884135,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":198.86809148141253,"max_completion_cost":24.858511435176567,"max_cost":198.86809148141253},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3.1-euryale-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":8.1235e-07,"completion":8.1235e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.1064763392,"max_completion_cost":0.0133095424,"max_cost":0.1064763392},"sats_pricing":{"prompt":0.0012494943420139877,"completion":0.0012494943420139877,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":163.7737223964574,"max_completion_cost":20.471715299557175,"max_cost":163.7737223964574},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","alias_ids":null,"forwarded_model_id":null},{"id":"hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":1.1605e-06,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.152109056,"max_completion_cost":0.019013632,"max_cost":0.152109056},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":233.96246056636767,"max_completion_cost":29.24530757079596,"max_cost":233.96246056636767},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","alias_ids":null,"forwarded_model_id":null},{"id":"l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.642e-08,"completion":5.8025000000000005e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00038027264,"max_completion_cost":0.00047534080000000004,"max_cost":0.00047534080000000004},"sats_pricing":{"prompt":7.139967668651358e-05,"completion":8.924959585814197e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5849061514159193,"max_completion_cost":0.731132689269899,"max_cost":0.731132689269899},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"sao10k/l3-lunaris-8b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-08-06","name":"OpenAI: GPT-4o (2024-08-06)","created":1722902400,"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3750000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.37136,"max_completion_cost":0.19013632000000003,"max_cost":0.5139622400000001},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0021149193331313266,"input_cache_write":0.0,"max_prompt_cost":571.1974134921087,"max_completion_cost":292.45307570795967,"max_cost":790.5372202730786},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-2024-08-06","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-70b-instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":4.6420000000000004e-07,"completion":4.6420000000000004e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.060843622400000005,"max_completion_cost":0.007605452800000001,"max_cost":0.060843622400000005},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.0007139967668651358,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":93.58498422654708,"max_completion_cost":11.698123028318385,"max_cost":93.58498422654708},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.1-70b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"llama-3.1-8b-instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":5.8025000000000005e-08,"completion":9.284e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.007605452800000001,"max_completion_cost":0.01216872448,"max_cost":0.01216872448},"sats_pricing":{"prompt":8.924959585814197e-05,"completion":0.00014279935337302717,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":4.2298386662626526e-05,"input_cache_write":0.0,"max_prompt_cost":11.698123028318385,"max_completion_cost":18.716996845309417,"max_cost":18.716996845309417},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":131072,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"meta-llama/llama-3.1-8b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-nemo","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.2049500000000005e-08,"completion":3.4815e-08,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0028900720640000007,"max_completion_cost":0.00057040896,"max_cost":0.0030992220160000004},"sats_pricing":{"prompt":3.3914846426093956e-05,"completion":5.354975751488518e-05,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.445286750760987,"max_completion_cost":0.8773592271238788,"max_cost":4.766985134039742},"per_request_limits":null,"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-nemo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini","name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.022281600000000002,"max_completion_cost":0.011408179200000002,"max_cost":0.030837734400000004},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012689515998787958,"input_cache_write":0.0,"max_prompt_cost":34.27184480952652,"max_completion_cost":17.54718454247758,"max_cost":47.4322332163847},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.7407500000000003e-07,"completion":6.963000000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":8.25e-08,"input_cache_write":0.0,"max_prompt_cost":0.022281600000000002,"max_completion_cost":0.011408179200000002,"max_cost":0.030837734400000004},"sats_pricing":{"prompt":0.000267748787574426,"completion":0.001070995150297704,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00012689515998787958,"input_cache_write":0.0,"max_prompt_cost":34.27184480952652,"max_completion_cost":17.54718454247758,"max_cost":47.4322332163847},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-mini-2024-07-18","alias_ids":null,"forwarded_model_id":null},{"id":"gemma-2-27b-it","name":"Google: Gemma 2 27B","created":1720828800,"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":7.54325e-07,"completion":7.54325e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0061794304,"max_completion_cost":0.0015448576,"max_cost":0.0061794304},"sats_pricing":{"prompt":0.0011602447461558456,"completion":0.0011602447461558456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.504724960508687,"max_completion_cost":2.376181240127172,"max_cost":9.504724960508687},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"google/gemma-2-27b-it","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o","name":"OpenAI: GPT-4o","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.9012500000000004e-06,"completion":1.1605000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":1.3750000000000002e-06,"input_cache_write":0.0,"max_prompt_cost":0.37136,"max_completion_cost":0.19013632000000003,"max_cost":0.5139622400000001},"sats_pricing":{"prompt":0.0044624797929070995,"completion":0.017849919171628398,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0021149193331313266,"input_cache_write":0.0,"max_prompt_cost":571.1974134921087,"max_completion_cost":292.45307570795967,"max_cost":790.5372202730786},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-2024-05-13","name":"OpenAI: GPT-4o (2024-05-13)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.802500000000001e-06,"completion":1.74075e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.74272,"max_completion_cost":0.07130112,"max_cost":0.7902540800000001},"sats_pricing":{"prompt":0.008924959585814199,"completion":0.02677487875744259,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1142.3948269842174,"max_completion_cost":109.66990339048485,"max_cost":1215.5080959112072},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4o-2024-05-13","alias_ids":null,"forwarded_model_id":null},{"id":"mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","created":1713312000,"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":2.321e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.152109056,"max_completion_cost":0.45632716800000006,"max_cost":0.45632716800000006},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":233.96246056636767,"max_completion_cost":701.8873816991031,"max_cost":701.8873816991031},"per_request_limits":null,"top_provider":{"context_length":65536,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mixtral-8x22b-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"wizardlm-2-8x22b","name":"WizardLM-2 8x22B","created":1713225600,"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65535,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":7.195100000000001e-07,"completion":7.195100000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.04715308785000001,"max_completion_cost":0.005756080000000001,"max_cost":0.04715308785000001},"sats_pricing":{"prompt":0.0011066949886409606,"completion":0.0011066949886409606,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":72.52725608058536,"max_completion_cost":8.853559909127686,"max_cost":72.52725608058536},"per_request_limits":null,"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"microsoft/wizardlm-2-8x22b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo","name":"OpenAI: GPT-4 Turbo","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.48544,"max_completion_cost":0.14260224,"max_cost":1.5805081600000002},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2284.789653968435,"max_completion_cost":219.3398067809697,"max_cost":2431.0161918224144},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"claude-3-haiku","name":"Anthropic: Claude 3 Haiku","created":1710288000,"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":2.90125e-07,"completion":1.4506250000000002e-06,"request":0.0,"image":0.0,"web_search":0.011000000000000001,"internal_reasoning":0.0,"input_cache_read":3.3e-08,"input_cache_write":3.3e-07,"max_prompt_cost":0.058025,"max_completion_cost":0.005941760000000001,"max_cost":0.062778408},"sats_pricing":{"prompt":0.0004462479792907098,"completion":0.0022312398964535497,"request":0.001,"image":0.0,"web_search":16.919354665050612,"internal_reasoning":0.0,"input_cache_read":5.075806399515183e-05,"input_cache_write":0.0005075806399515183,"max_prompt_cost":89.24959585814197,"max_completion_cost":9.13915861587374,"max_cost":96.56092275084094},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"anthropic/claude-3-haiku","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-large","name":"Mistral Large","created":1708905600,"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":2.321e-06,"completion":6.963000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2e-07,"input_cache_write":0.0,"max_prompt_cost":0.29708799999999996,"max_completion_cost":0.8912640000000002,"max_cost":0.8912640000000002},"sats_pricing":{"prompt":0.0035699838343256785,"completion":0.010709951502977037,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0003383870933010122,"input_cache_write":0.0,"max_prompt_cost":456.95793079368684,"max_completion_cost":1370.8737923810609,"max_cost":1370.8737923810609},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mistralai/mistral-large","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","created":1706140800,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605e-06,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0047522475,"max_completion_cost":0.009504495,"max_cost":0.009504495},"sats_pricing":{"prompt":0.0017849919171628393,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.309541900781827,"max_completion_cost":14.619083801563654,"max_cost":14.619083801563654},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo-0613","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview","created":1706140800,"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.1605000000000002e-05,"completion":3.4815e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.48544,"max_completion_cost":0.14260224,"max_cost":1.5805081600000002},"sats_pricing":{"prompt":0.017849919171628398,"completion":0.05354975751488518,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2284.789653968435,"max_completion_cost":219.3398067809697,"max_cost":2431.0161918224144},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4-turbo-preview","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","created":1695859200,"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":1.7407500000000002e-06,"completion":2.321e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.007128371250000001,"max_completion_cost":0.009504495,"max_cost":0.009504495},"sats_pricing":{"prompt":0.0026774878757442593,"completion":0.0035699838343256785,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":10.964312851172743,"max_completion_cost":14.619083801563654,"max_cost":14.619083801563654},"per_request_limits":null,"top_provider":{"context_length":4095,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo-instruct","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","created":1693180800,"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.4815000000000005e-06,"completion":4.642e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.05704437750000001,"max_completion_cost":0.019013632,"max_cost":0.06179778550000001},"sats_pricing":{"prompt":0.005354975751488519,"completion":0.007139967668651357,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":87.74127768813938,"max_completion_cost":29.24530757079596,"max_cost":95.05260458083838},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo-16k","alias_ids":null,"forwarded_model_id":null},{"id":"weaver","name":"Mancer: Weaver (alpha)","created":1690934400,"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":5.8025e-07,"completion":8.703750000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.004641999999999999,"max_completion_cost":0.00522225,"max_cost":0.00638275},"sats_pricing":{"prompt":0.0008924959585814196,"completion":0.0013387439378721297,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":7.139967668651357,"max_completion_cost":8.032463627232778,"max_cost":9.817455544395617},"per_request_limits":null,"top_provider":{"context_length":8000,"max_completion_tokens":6000,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"mancer/weaver","alias_ids":null,"forwarded_model_id":null},{"id":"remm-slerp-l2-13b","name":"ReMM SLERP 13B","created":1689984000,"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":5.22225e-07,"completion":7.54325e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0032085504000000003,"max_completion_cost":0.0046345728,"max_cost":0.0046345728},"sats_pricing":{"prompt":0.0008032463627232777,"completion":0.0011602447461558456,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.935145652571819,"max_completion_cost":7.128543720381516,"max_cost":7.128543720381516},"per_request_limits":null,"top_provider":{"context_length":6144,"max_completion_tokens":6144,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"undi95/remm-slerp-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"mythomax-l2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":9.284e-08,"completion":1.2765500000000001e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00038027264,"max_completion_cost":0.0005228748800000001,"max_cost":0.0005228748800000001},"sats_pricing":{"prompt":0.00014279935337302717,"completion":0.00019634911088791236,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.5849061514159193,"max_completion_cost":0.804245958196889,"max_cost":0.804245958196889},"per_request_limits":null,"top_provider":{"context_length":4096,"max_completion_tokens":4096,"is_moderated":false},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"gryphe/mythomax-l2-13b","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo","name":"OpenAI: GPT-3.5 Turbo","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.8025e-07,"completion":1.7407500000000002e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00950739625,"max_completion_cost":0.007130112000000001,"max_cost":0.014260804250000002},"sats_pricing":{"prompt":0.0008924959585814196,"completion":0.0026774878757442593,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":14.62354628135656,"max_completion_cost":10.966990339048486,"max_cost":21.934873174055554},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4","name":"OpenAI: GPT-4","created":1685232000,"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.4815e-05,"completion":6.963e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.28516966499999996,"max_completion_cost":0.28520448,"max_cost":0.42777190499999995},"sats_pricing":{"prompt":0.05354975751488518,"completion":0.10709951502977036,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":438.62606380442446,"max_completion_cost":438.6796135619394,"max_cost":657.9658705853942},"per_request_limits":null,"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":"openai/gpt-4","alias_ids":null,"forwarded_model_id":null},{"id":"autoclaw","name":"🔥 AutoClaw (for OpenClaw)","created":1786172188,"description":"PPQ.AI model","context_length":200000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":-1.1e-06,"completion":-1.1e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-0.22,"max_completion_cost":-0.22,"max_cost":-0.22},"sats_pricing":{"prompt":-0.0016919354665050612,"completion":-0.0016919354665050612,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":-338.3870933010122,"max_completion_cost":-338.3870933010122,"max_cost":0.001},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"llama3-3-70b","name":"Llama 3.3 70B (Private via TEE)","created":1786172188,"description":"PPQ.AI model","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":2.030875e-06,"completion":3.1913749999999996e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.266190848,"max_completion_cost":0.41829990399999994,"max_cost":0.41829990399999994},"sats_pricing":{"prompt":0.003123735855034969,"completion":0.004908727772197807,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":409.43430599114345,"max_completion_cost":643.396766557511,"max_cost":643.396766557511},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-vl-30b","name":"Qwen3-VL 30B (Private via TEE)","created":1786172188,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.450625e-06,"completion":4.642e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.38027264,"max_completion_cost":1.216872448,"max_cost":1.216872448},"sats_pricing":{"prompt":0.0022312398964535493,"completion":0.007139967668651357,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":584.9061514159192,"max_completion_cost":1871.6996845309413,"max_cost":1871.6996845309413},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"glm-5-2","name":"GLM-5.2 (Private via TEE)","created":1786172188,"description":"PPQ.AI model","context_length":384000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.7407500000000002e-06,"completion":6.092625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.668448,"max_completion_cost":2.339568,"max_cost":2.339568},"sats_pricing":{"prompt":0.0026774878757442593,"completion":0.009371207565104907,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1028.1553442857955,"max_completion_cost":3598.5437050002843,"max_cost":3598.5437050002843},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"gemma4-31b","name":"Gemma 4 31B (Private via TEE)","created":1786172188,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":4.6420000000000004e-07,"completion":1.1605e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.12168724480000001,"max_completion_cost":0.304218112,"max_cost":0.304218112},"sats_pricing":{"prompt":0.0007139967668651358,"completion":0.0017849919171628393,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":187.16996845309416,"max_completion_cost":467.92492113273534,"max_cost":467.92492113273534},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2-6","name":"Kimi K2.6 (Private via TEE)","created":1786172188,"description":"PPQ.AI model","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":1.7407500000000002e-06,"completion":6.092625e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.45632716800000006,"max_completion_cost":1.597145088,"max_cost":1.597145088},"sats_pricing":{"prompt":0.0026774878757442593,"completion":0.009371207565104907,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":701.8873816991031,"max_completion_cost":2456.6058359468607,"max_cost":2456.6058359468607},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2-fast","name":"GLM 5.2 (Fast)","created":1786172188,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":2.4370500000000002e-06,"completion":7.659300000000001e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.5554321408000003,"max_completion_cost":8.031358156800001,"max_cost":8.031358156800001},"sats_pricing":{"prompt":0.0037484830260419632,"completion":0.011780946653274742,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3930.5693375149776,"max_completion_cost":12353.217917904216,"max_cost":12353.217917904216},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k3-fast","name":"Kimi K3 (Fast)","created":1786172188,"description":"PPQ.AI model","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Unknown","instruct_type":null},"pricing":{"prompt":5.22225e-06,"completion":2.6111250000000002e-05,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":5.475926016,"max_completion_cost":27.379630080000002,"max_cost":27.379630080000002},"sats_pricing":{"prompt":0.008032463627232778,"completion":0.04016231813616389,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":8422.648580389237,"max_completion_cost":42113.242901946185,"max_cost":42113.242901946185},"per_request_limits":null,"top_provider":null,"enabled":true,"upstream_provider_id":"ppqai","canonical_slug":null,"alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-5:batch","name":"Claude Opus 5 (batch)","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":9.375000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":2.3437500000000002e-06,"max_prompt_cost":1.8750000000000002,"max_completion_cost":1.2000000000000002,"max_cost":2.8350000000000004},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.014419904544077227,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0036049761360193067,"max_prompt_cost":2883.980908815445,"max_completion_cost":1845.747781641885,"max_cost":4360.579134128953},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-opus-5-20260723","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.6-flash:batch","name":"Google: Gemini 3.6 Flash (batch)","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":2.8125e-06,"request":0.0,"image":5.625e-07,"web_search":0.0105,"internal_reasoning":2.8125e-06,"input_cache_read":5.625e-08,"input_cache_write":6.249999999999997e-08,"max_prompt_cost":0.589824,"max_completion_cost":0.18432,"max_cost":0.73728},"sats_pricing":{"prompt":0.0008651942726446336,"completion":0.004325971363223167,"request":0.001,"image":0.0008651942726446336,"web_search":16.150293089366492,"internal_reasoning":0.004325971363223167,"input_cache_read":8.651942726446334e-05,"input_cache_write":9.61326969605148e-05,"max_prompt_cost":907.2219496326193,"max_completion_cost":283.5068592601935,"max_cost":1134.027437040774},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.6-flash-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash-lite:batch","name":"Google: Gemini 3.5 Flash Lite (batch)","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":9.375000000000001e-07,"request":0.0,"image":1.125e-07,"web_search":0.0105,"internal_reasoning":9.375000000000001e-07,"input_cache_read":1.1249999999999998e-08,"input_cache_write":0.0,"max_prompt_cost":0.1179648,"max_completion_cost":0.06144000000000001,"max_cost":0.17203200000000002},"sats_pricing":{"prompt":0.00017303885452892668,"completion":0.0014419904544077226,"request":0.001,"image":0.00017303885452892668,"web_search":16.150293089366492,"internal_reasoning":0.0014419904544077226,"input_cache_read":1.730388545289267e-05,"input_cache_write":0.0,"max_prompt_cost":181.44438992652383,"max_completion_cost":94.50228642006451,"max_cost":264.6064019761806},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-lite-20260721","alias_ids":null,"forwarded_model_id":null},{"id":"inkling:batch","name":"Thinking Machines: Inkling (batch)","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.75e-07,"completion":1.51875e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":6.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.196608,"max_completion_cost":0.7962624,"max_cost":0.7962624},"sats_pricing":{"prompt":0.000576796181763089,"completion":0.0023360245361405104,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.805535089972513e-05,"input_cache_write":0.0,"max_prompt_cost":302.4073165442064,"max_completion_cost":1224.749632004036,"max_cost":1224.749632004036},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thinkingmachines/inkling-20260715","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna-pro:batch","name":"OpenAI: GPT-5.6 Luna Pro (batch)","created":1783590867,"description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-08,"completion":4.5e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":7.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.07875,"max_completion_cost":0.0576,"max_cost":0.12675},"sats_pricing":{"prompt":0.0001153592363526178,"completion":0.0006921554181157067,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":1.1535923635261782e-05,"input_cache_write":0.0,"max_prompt_cost":121.12719817024869,"max_completion_cost":88.59589351881047,"max_cost":194.95710943592408},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-luna:batch","name":"OpenAI: GPT-5.6 Luna (batch)","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-08,"completion":4.5e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":7.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.07875,"max_completion_cost":0.0576,"max_cost":0.12675},"sats_pricing":{"prompt":0.0001153592363526178,"completion":0.0006921554181157067,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":1.1535923635261782e-05,"input_cache_write":0.0,"max_prompt_cost":121.12719817024869,"max_completion_cost":88.59589351881047,"max_cost":194.95710943592408},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-luna-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra-pro:batch","name":"OpenAI: GPT-5.6 Terra Pro (batch)","created":1783590861,"description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-07,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":7.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.7875,"max_completion_cost":0.5760000000000001,"max_cost":1.2675},"sats_pricing":{"prompt":0.001153592363526178,"completion":0.006921554181157068,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0001153592363526178,"input_cache_write":0.0,"max_prompt_cost":1211.2719817024868,"max_completion_cost":885.9589351881048,"max_cost":1949.571094359241},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-terra:batch","name":"OpenAI: GPT-5.6 Terra (batch)","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-07,"completion":4.5e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":7.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.7875,"max_completion_cost":0.5760000000000001,"max_cost":1.2675},"sats_pricing":{"prompt":0.001153592363526178,"completion":0.006921554181157068,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0001153592363526178,"input_cache_write":0.0,"max_prompt_cost":1211.2719817024868,"max_completion_cost":885.9589351881048,"max_cost":1949.571094359241},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-terra-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol-pro:batch","name":"OpenAI: GPT-5.6 Sol Pro (batch)","created":1783590854,"description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks.\n\nLearn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":1.125e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":0.0,"max_prompt_cost":1.9687500000000002,"max_completion_cost":1.4400000000000002,"max_cost":3.16875},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.01730388545289267,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0,"max_prompt_cost":3028.1799542562176,"max_completion_cost":2214.897337970262,"max_cost":4873.927735898103},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.6-sol:batch","name":"OpenAI: GPT-5.6 Sol (batch)","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":1.125e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":0.0,"max_prompt_cost":1.9687500000000002,"max_completion_cost":1.4400000000000002,"max_cost":3.16875},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.01730388545289267,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0,"max_prompt_cost":3028.1799542562176,"max_completion_cost":2214.897337970262,"max_cost":4873.927735898103},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.6-sol-20260709","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-5:batch","name":"Anthropic: Claude Sonnet 5 (batch)","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":7.5e-07,"completion":3.7500000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":7.5e-08,"input_cache_write":9.375000000000001e-07,"max_prompt_cost":0.75,"max_completion_cost":0.4800000000000001,"max_cost":1.1340000000000001},"sats_pricing":{"prompt":0.001153592363526178,"completion":0.00576796181763089,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0001153592363526178,"input_cache_write":0.0014419904544077226,"max_prompt_cost":1153.592363526178,"max_completion_cost":738.299112656754,"max_cost":1744.2316536515814},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-sonnet-5-20260630","alias_ids":null,"forwarded_model_id":null},{"id":"glm-5.2:batch","name":"Z.ai: GLM 5.2 (batch)","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":512000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":5.25e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":9.749999999999999e-08,"input_cache_write":0.0,"max_prompt_cost":0.2688,"max_completion_cost":0.8448,"max_cost":0.8448},"sats_pricing":{"prompt":0.0008075146544683245,"completion":0.0025379031997575918,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00014996700725840312,"input_cache_write":0.0,"max_prompt_cost":413.4475030877822,"max_completion_cost":1299.4064382758868,"max_cost":1299.4064382758868},"per_request_limits":null,"top_provider":{"context_length":512000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"z-ai/glm-5.2-20260616","alias_ids":null,"forwarded_model_id":null},{"id":"kimi-k2.7-code:batch","name":"MoonshotAI: Kimi K2.7 Code (batch)","created":1781266361,"description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.5625e-07,"completion":1.5e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.125e-08,"input_cache_write":0.0,"max_prompt_cost":0.0933888,"max_completion_cost":0.393216,"max_cost":0.393216},"sats_pricing":{"prompt":0.0005479563726749345,"completion":0.002307184727052356,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.00010959127453498692,"input_cache_write":0.0,"max_prompt_cost":143.64347535849802,"max_completion_cost":604.8146330884128,"max_cost":604.8146330884128},"per_request_limits":null,"top_provider":{"context_length":262144,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","alias_ids":null,"forwarded_model_id":null},{"id":"claude-fable-5:batch","name":"Anthropic: Claude Fable 5 (batch)","created":1781007515,"description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-06,"completion":1.8750000000000002e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":3.75e-07,"input_cache_write":4.6875000000000004e-06,"max_prompt_cost":3.7500000000000004,"max_completion_cost":2.4000000000000004,"max_cost":5.670000000000001},"sats_pricing":{"prompt":0.00576796181763089,"completion":0.028839809088154453,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.000576796181763089,"input_cache_write":0.007209952272038613,"max_prompt_cost":5767.96181763089,"max_completion_cost":3691.49556328377,"max_cost":8721.158268257906},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-5-fable-20260609","alias_ids":null,"forwarded_model_id":null},{"id":"nemotron-3-ultra-550b-a55b:batch","name":"NVIDIA: Nemotron 3 Ultra (batch)","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":512288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.25e-07,"completion":1.35e-06,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":7.5e-08,"input_cache_write":0.0,"max_prompt_cost":0.1152648,"max_completion_cost":0.6915888,"max_cost":0.6915888},"sats_pricing":{"prompt":0.00034607770905785337,"completion":0.0020764662543471205,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0001153592363526178,"input_cache_write":0.0,"max_prompt_cost":177.2914574178296,"max_completion_cost":1063.7487445069776,"max_cost":1063.7487445069776},"per_request_limits":null,"top_provider":{"context_length":512288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","alias_ids":null,"forwarded_model_id":null},{"id":"minimax-m3:batch","name":"MiniMax: MiniMax M3 (batch)","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":524288,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":4.5e-07,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":2.2499999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.0589824,"max_completion_cost":0.2359296,"max_cost":0.2359296},"sats_pricing":{"prompt":0.00017303885452892668,"completion":0.0006921554181157067,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":3.460777090578534e-05,"input_cache_write":0.0,"max_prompt_cost":90.72219496326191,"max_completion_cost":362.88877985304765,"max_cost":362.88877985304765},"per_request_limits":null,"top_provider":{"context_length":524288,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"minimax/minimax-m3-20260531","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.8:batch","name":"Anthropic: Claude Opus 4.8 (batch)","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":9.375000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":2.3437500000000002e-06,"max_prompt_cost":1.8750000000000002,"max_completion_cost":1.2000000000000002,"max_cost":2.8350000000000004},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.014419904544077227,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0036049761360193067,"max_prompt_cost":2883.980908815445,"max_completion_cost":1845.747781641885,"max_cost":4360.579134128953},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.8-opus-20260528","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.5-flash:batch","name":"Google: Gemini 3.5 Flash (batch)","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":5.625e-07,"completion":3.375e-06,"request":0.0,"image":5.625e-07,"web_search":0.0105,"internal_reasoning":3.375e-06,"input_cache_read":5.625e-08,"input_cache_write":0.0,"max_prompt_cost":0.589824,"max_completion_cost":0.221184,"max_cost":0.7741439999999999},"sats_pricing":{"prompt":0.0008651942726446336,"completion":0.0051911656358678004,"request":0.001,"image":0.0008651942726446336,"web_search":16.150293089366492,"internal_reasoning":0.0051911656358678004,"input_cache_read":8.651942726446334e-05,"input_cache_write":0.0,"max_prompt_cost":907.2219496326193,"max_completion_cost":340.20823111223217,"max_cost":1190.7288088928126},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.5-flash-20260519","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-flash-lite:batch","name":"Google: Gemini 3.1 Flash Lite (batch)","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":9.375e-08,"completion":5.625e-07,"request":0.0,"image":9.375e-08,"web_search":0.0105,"internal_reasoning":5.625e-07,"input_cache_read":9.375e-09,"input_cache_write":0.0,"max_prompt_cost":0.098304,"max_completion_cost":0.036864,"max_cost":0.129024},"sats_pricing":{"prompt":0.00014419904544077226,"completion":0.0008651942726446336,"request":0.001,"image":0.00014419904544077226,"web_search":16.150293089366492,"internal_reasoning":0.0008651942726446336,"input_cache_read":1.4419904544077224e-05,"input_cache_write":0.0,"max_prompt_cost":151.2036582721032,"max_completion_cost":56.701371852038704,"max_cost":198.45480148213545},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-flash-lite-20260507","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5-pro:batch","name":"OpenAI: GPT-5.5 Pro (batch)","created":1777051896,"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-05,"completion":6.75e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.8125,"max_completion_cost":8.64,"max_cost":19.012500000000003},"sats_pricing":{"prompt":0.01730388545289267,"completion":0.10382331271735602,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18169.079725537304,"max_completion_cost":13289.384027821572,"max_cost":29243.566415388617},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.5-pro-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.5:batch","name":"OpenAI: GPT-5.5 (batch)","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":1.125e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":0.0,"max_prompt_cost":1.9687500000000002,"max_completion_cost":1.4400000000000002,"max_cost":3.16875},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.01730388545289267,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0,"max_prompt_cost":3028.1799542562176,"max_completion_cost":2214.897337970262,"max_cost":4873.927735898103},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.5-20260423","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.7:batch","name":"Anthropic: Claude Opus 4.7 (batch)","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":9.375000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":2.3437500000000002e-06,"max_prompt_cost":1.8750000000000002,"max_completion_cost":1.2000000000000002,"max_cost":2.8350000000000004},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.014419904544077227,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0036049761360193067,"max_prompt_cost":2883.980908815445,"max_completion_cost":1845.747781641885,"max_cost":4360.579134128953},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.7-opus-20260416","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-nano:batch","name":"OpenAI: GPT-5.4 Nano (batch)","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-08,"completion":4.6875000000000006e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":7.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.03,"max_completion_cost":0.06000000000000001,"max_cost":0.08040000000000001},"sats_pricing":{"prompt":0.0001153592363526178,"completion":0.0007209952272038613,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":1.1535923635261782e-05,"input_cache_write":0.0,"max_prompt_cost":46.14369454104712,"max_completion_cost":92.28738908209425,"max_cost":123.6651013700063},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-nano-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-mini:batch","name":"OpenAI: GPT-5.4 Mini (batch)","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":2.8125e-07,"completion":1.6875e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":2.8125e-08,"input_cache_write":0.0,"max_prompt_cost":0.1125,"max_completion_cost":0.216,"max_cost":0.2925},"sats_pricing":{"prompt":0.0004325971363223168,"completion":0.0025955828179339002,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":4.325971363223167e-05,"input_cache_write":0.0,"max_prompt_cost":173.0388545289267,"max_completion_cost":332.23460069553926,"max_cost":449.9010217752094},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-mini-20260317","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4-pro:batch","name":"OpenAI: GPT-5.4 Pro (batch)","created":1772734366,"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.125e-05,"completion":6.75e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.8125,"max_completion_cost":8.64,"max_cost":19.012500000000003},"sats_pricing":{"prompt":0.01730388545289267,"completion":0.10382331271735602,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":18169.079725537304,"max_completion_cost":13289.384027821572,"max_cost":29243.566415388617},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-pro-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.4:batch","name":"OpenAI: GPT-5.4 (batch)","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.375000000000001e-07,"completion":5.625e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":9.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.9843750000000001,"max_completion_cost":0.7200000000000001,"max_cost":1.584375},"sats_pricing":{"prompt":0.0014419904544077226,"completion":0.008651942726446335,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00014419904544077226,"input_cache_write":0.0,"max_prompt_cost":1514.0899771281088,"max_completion_cost":1107.448668985131,"max_cost":2436.9638679490513},"per_request_limits":null,"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.4-20260305","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3.1-pro-preview:batch","name":"Google: Gemini 3.1 Pro Preview (batch)","created":1771509627,"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":7.5e-07,"completion":4.5e-06,"request":0.0,"image":7.5e-07,"web_search":0.0105,"internal_reasoning":4.5e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.786432,"max_completion_cost":0.294912,"max_cost":1.032192},"sats_pricing":{"prompt":0.001153592363526178,"completion":0.006921554181157068,"request":0.001,"image":0.001153592363526178,"web_search":16.150293089366492,"internal_reasoning":0.006921554181157068,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1209.6292661768257,"max_completion_cost":453.61097481630964,"max_cost":1587.6384118570836},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3.1-pro-preview-20260219","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.6:batch","name":"Anthropic: Claude Sonnet 4.6 (batch)","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":5.625e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":1.40625e-06,"max_prompt_cost":1.125,"max_completion_cost":0.7200000000000001,"max_cost":1.701},"sats_pricing":{"prompt":0.001730388545289267,"completion":0.008651942726446335,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00017303885452892668,"input_cache_write":0.0021629856816115837,"max_prompt_cost":1730.388545289267,"max_completion_cost":1107.448668985131,"max_cost":2616.347480477372},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.6:batch","name":"Anthropic: Claude Opus 4.6 (batch)","created":1770219050,"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":9.375000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":2.3437500000000002e-06,"max_prompt_cost":1.8750000000000002,"max_completion_cost":1.2000000000000002,"max_cost":2.8350000000000004},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.014419904544077227,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0036049761360193067,"max_prompt_cost":2883.980908815445,"max_completion_cost":1845.747781641885,"max_cost":4360.579134128953},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.6-opus-20260205","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-3-flash-preview:batch","name":"Google: Gemini 3 Flash Preview (batch)","created":1765987078,"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.875e-07,"completion":1.125e-06,"request":0.0,"image":1.875e-07,"web_search":0.0105,"internal_reasoning":1.125e-06,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.196608,"max_completion_cost":0.073728,"max_cost":0.258048},"sats_pricing":{"prompt":0.0002883980908815445,"completion":0.001730388545289267,"request":0.001,"image":0.0002883980908815445,"web_search":16.150293089366492,"internal_reasoning":0.001730388545289267,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":302.4073165442064,"max_completion_cost":113.40274370407741,"max_cost":396.9096029642709},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-3-flash-preview-20251217","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2-pro:batch","name":"OpenAI: GPT-5.2 Pro (batch)","created":1765389780,"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.875e-06,"completion":6.3e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.15,"max_completion_cost":8.064,"max_cost":10.206},"sats_pricing":{"prompt":0.012112719817024869,"completion":0.09690175853619895,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4845.087926809947,"max_completion_cost":12403.425092633466,"max_cost":15698.08488286423},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-pro-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.2:batch","name":"OpenAI: GPT-5.2 (batch)","created":1765389775,"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":6.5625e-07,"completion":5.25e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":6.5625e-08,"input_cache_write":0.0,"max_prompt_cost":0.2625,"max_completion_cost":0.6719999999999999,"max_cost":0.8504999999999999},"sats_pricing":{"prompt":0.0010093933180854056,"completion":0.008075146544683245,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00010093933180854057,"input_cache_write":0.0,"max_prompt_cost":403.7573272341623,"max_completion_cost":1033.6187577194553,"max_cost":1308.1737402386857},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.2-20251211","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.5:batch","name":"Anthropic: Claude Opus 4.5 (batch)","created":1764010580,"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.8750000000000003e-06,"completion":9.375000000000001e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":2.3437500000000002e-06,"max_prompt_cost":0.37500000000000006,"max_completion_cost":0.6000000000000001,"max_cost":0.8550000000000002},"sats_pricing":{"prompt":0.002883980908815445,"completion":0.014419904544077227,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0036049761360193067,"max_prompt_cost":576.7961817630891,"max_completion_cost":922.8738908209425,"max_cost":1315.0952944198432},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-opus-20251124","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5.1:batch","name":"OpenAI: GPT-5.1 (batch)","created":1763060305,"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.6875000000000006e-07,"completion":3.7500000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":4.6875e-08,"input_cache_write":0.0,"max_prompt_cost":0.18750000000000003,"max_completion_cost":0.4800000000000001,"max_cost":0.6075000000000002},"sats_pricing":{"prompt":0.0007209952272038613,"completion":0.00576796181763089,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":7.209952272038613e-05,"input_cache_write":0.0,"max_prompt_cost":288.39809088154453,"max_completion_cost":738.299112656754,"max_cost":934.4098144562045},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5.1-20251113","alias_ids":null,"forwarded_model_id":null},{"id":"claude-haiku-4.5:batch","name":"Anthropic: Claude Haiku 4.5 (batch)","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":3.75e-07,"completion":1.8750000000000003e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":3.75e-08,"input_cache_write":4.6875000000000006e-07,"max_prompt_cost":0.075,"max_completion_cost":0.12000000000000002,"max_cost":0.17100000000000004},"sats_pricing":{"prompt":0.000576796181763089,"completion":0.002883980908815445,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":5.76796181763089e-05,"input_cache_write":0.0007209952272038613,"max_prompt_cost":115.3592363526178,"max_completion_cost":184.5747781641885,"max_cost":263.0190588839686},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-haiku-20251001","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-pro:batch","name":"OpenAI: GPT-5 Pro (batch)","created":1759776663,"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-06,"completion":4.5e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.25,"max_completion_cost":5.760000000000001,"max_cost":7.290000000000001},"sats_pricing":{"prompt":0.008651942726446335,"completion":0.06921554181157068,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3460.777090578534,"max_completion_cost":8859.589351881048,"max_cost":11212.917773474452},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-pro-2025-10-06","alias_ids":null,"forwarded_model_id":null},{"id":"claude-sonnet-4.5:batch","name":"Anthropic: Claude Sonnet 4.5 (batch)","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":1.125e-06,"completion":5.625e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.125e-07,"input_cache_write":1.40625e-06,"max_prompt_cost":1.125,"max_completion_cost":0.36000000000000004,"max_cost":1.413},"sats_pricing":{"prompt":0.001730388545289267,"completion":0.008651942726446335,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00017303885452892668,"input_cache_write":0.0021629856816115837,"max_prompt_cost":1730.388545289267,"max_completion_cost":553.7243344925655,"max_cost":2173.3680128833194},"per_request_limits":null,"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-codex:batch","name":"OpenAI: GPT-5 Codex (batch)","created":1758643403,"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.6875000000000006e-07,"completion":3.7500000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":4.6875e-08,"input_cache_write":0.0,"max_prompt_cost":0.18750000000000003,"max_completion_cost":0.4800000000000001,"max_cost":0.6075000000000002},"sats_pricing":{"prompt":0.0007209952272038613,"completion":0.00576796181763089,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":7.209952272038613e-05,"input_cache_write":0.0,"max_prompt_cost":288.39809088154453,"max_completion_cost":738.299112656754,"max_cost":934.4098144562045},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-codex","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5:batch","name":"OpenAI: GPT-5 (batch)","created":1754587413,"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.6875000000000006e-07,"completion":3.7500000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":4.6875e-08,"input_cache_write":0.0,"max_prompt_cost":0.18750000000000003,"max_completion_cost":0.4800000000000001,"max_cost":0.6075000000000002},"sats_pricing":{"prompt":0.0007209952272038613,"completion":0.00576796181763089,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":7.209952272038613e-05,"input_cache_write":0.0,"max_prompt_cost":288.39809088154453,"max_completion_cost":738.299112656754,"max_cost":934.4098144562045},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-mini:batch","name":"OpenAI: GPT-5 Mini (batch)","created":1754587407,"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.375e-08,"completion":7.5e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":9.375e-09,"input_cache_write":0.0,"max_prompt_cost":0.0375,"max_completion_cost":0.096,"max_cost":0.1215},"sats_pricing":{"prompt":0.00014419904544077226,"completion":0.001153592363526178,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":1.4419904544077224e-05,"input_cache_write":0.0,"max_prompt_cost":57.6796181763089,"max_completion_cost":147.65982253135078,"max_cost":186.88196289124082},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-mini-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-5-nano:batch","name":"OpenAI: GPT-5 Nano (batch)","created":1754587402,"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.875e-08,"completion":1.5e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.8750000000000002e-09,"input_cache_write":0.0,"max_prompt_cost":0.0075,"max_completion_cost":0.0192,"max_cost":0.0243},"sats_pricing":{"prompt":2.883980908815445e-05,"completion":0.0002307184727052356,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":2.8839809088154454e-06,"input_cache_write":0.0,"max_prompt_cost":11.53592363526178,"max_completion_cost":29.531964506270153,"max_cost":37.376392578248165},"per_request_limits":null,"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-5-nano-2025-08-07","alias_ids":null,"forwarded_model_id":null},{"id":"claude-opus-4.1:batch","name":"Anthropic: Claude Opus 4.1 (batch)","created":1754411591,"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":5.625e-06,"completion":2.8125e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":5.625e-07,"input_cache_write":7.03125e-06,"max_prompt_cost":1.125,"max_completion_cost":0.9,"max_cost":1.8450000000000002},"sats_pricing":{"prompt":0.008651942726446335,"completion":0.043259713632231675,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0008651942726446336,"input_cache_write":0.010814928408057919,"max_prompt_cost":1730.388545289267,"max_completion_cost":1384.3108362314135,"max_cost":2837.8372142743983},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"anthropic/claude-4.1-opus-20250805","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash-lite:batch","name":"Google: Gemini 2.5 Flash Lite (batch)","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":3.75e-08,"completion":1.5e-07,"request":0.0,"image":3.75e-08,"web_search":0.0105,"internal_reasoning":1.5e-07,"input_cache_read":7.500000000000001e-09,"input_cache_write":0.0,"max_prompt_cost":0.0393216,"max_completion_cost":0.009830249999999999,"max_cost":0.046694287499999994},"sats_pricing":{"prompt":5.76796181763089e-05,"completion":0.0002307184727052356,"request":0.001,"image":5.76796181763089e-05,"web_search":16.150293089366492,"internal_reasoning":0.0002307184727052356,"input_cache_read":1.1535923635261782e-05,"input_cache_write":0.0,"max_prompt_cost":60.48146330884128,"max_completion_cost":15.120135108737612,"max_cost":71.82156464039448},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash-lite","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-flash:batch","name":"Google: Gemini 2.5 Flash (batch)","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":9.375000000000001e-07,"request":0.0,"image":1.125e-07,"web_search":0.0105,"internal_reasoning":9.375000000000001e-07,"input_cache_read":2.2499999999999996e-08,"input_cache_write":0.0,"max_prompt_cost":0.1179648,"max_completion_cost":0.06143906250000001,"max_cost":0.172031175},"sats_pricing":{"prompt":0.00017303885452892668,"completion":0.0014419904544077226,"request":0.001,"image":0.00017303885452892668,"web_search":16.150293089366492,"internal_reasoning":0.0014419904544077226,"input_cache_read":3.460777090578534e-05,"input_cache_write":0.0,"max_prompt_cost":181.44438992652383,"max_completion_cost":94.50084442961011,"max_cost":264.6051330245807},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-flash","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-2.5-pro:batch","name":"Google: Gemini 2.5 Pro (batch)","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":4.6875000000000006e-07,"completion":3.7500000000000005e-06,"request":0.0,"image":4.6875000000000006e-07,"web_search":0.0105,"internal_reasoning":3.7500000000000005e-06,"input_cache_read":9.375e-08,"input_cache_write":0.0,"max_prompt_cost":0.49152000000000007,"max_completion_cost":0.24576000000000003,"max_cost":0.7065600000000001},"sats_pricing":{"prompt":0.0007209952272038613,"completion":0.00576796181763089,"request":0.001,"image":0.0007209952272038613,"web_search":16.150293089366492,"internal_reasoning":0.00576796181763089,"input_cache_read":0.00014419904544077226,"input_cache_write":0.0,"max_prompt_cost":756.0182913605161,"max_completion_cost":378.00914568025803,"max_cost":1086.776293830742},"per_request_limits":null,"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-2.5-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-pro:batch","name":"OpenAI: o3 Pro (batch)","created":1749598352,"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.500000000000001e-06,"completion":3.0000000000000004e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.5000000000000002,"max_completion_cost":3.0000000000000004,"max_cost":3.7500000000000004},"sats_pricing":{"prompt":0.01153592363526178,"completion":0.04614369454104712,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2307.1847270523563,"max_completion_cost":4614.3694541047125,"max_cost":5767.96181763089},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-pro-2025-06-10","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini-high:batch","name":"OpenAI: o4 Mini High (batch)","created":1744824212,"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.03125e-07,"input_cache_write":0.0,"max_prompt_cost":0.0825,"max_completion_cost":0.165,"max_cost":0.20625000000000002},"sats_pricing":{"prompt":0.0006344757999393979,"completion":0.0025379031997575918,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00015861894998484948,"input_cache_write":0.0,"max_prompt_cost":126.89515998787958,"max_completion_cost":253.79031997575916,"max_cost":317.23789996969896},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o4-mini-high-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o3:batch","name":"OpenAI: o3 (batch)","created":1744823457,"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-07,"completion":3e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":0.0,"max_prompt_cost":0.15,"max_completion_cost":0.3,"max_cost":0.375},"sats_pricing":{"prompt":0.001153592363526178,"completion":0.004614369454104712,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0,"max_prompt_cost":230.7184727052356,"max_completion_cost":461.4369454104712,"max_cost":576.796181763089},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"o4-mini:batch","name":"OpenAI: o4 Mini (batch)","created":1744820942,"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.03125e-07,"input_cache_write":0.0,"max_prompt_cost":0.0825,"max_completion_cost":0.165,"max_cost":0.20625000000000002},"sats_pricing":{"prompt":0.0006344757999393979,"completion":0.0025379031997575918,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00015861894998484948,"input_cache_write":0.0,"max_prompt_cost":126.89515998787958,"max_completion_cost":253.79031997575916,"max_cost":317.23789996969896},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o4-mini-2025-04-16","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1:batch","name":"OpenAI: GPT-4.1 (batch)","created":1744651385,"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":7.5e-07,"completion":3e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":1.875e-07,"input_cache_write":0.0,"max_prompt_cost":0.785682,"max_completion_cost":0.098304,"max_cost":0.85941},"sats_pricing":{"prompt":0.001153592363526178,"completion":0.004614369454104712,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0002883980908815445,"input_cache_write":0.0,"max_prompt_cost":1208.4756738132994,"max_completion_cost":151.2036582721032,"max_cost":1321.8784175173769},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-mini:batch","name":"OpenAI: GPT-4.1 Mini (batch)","created":1744651381,"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.5e-07,"completion":6e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":3.75e-08,"input_cache_write":0.0,"max_prompt_cost":0.15713639999999998,"max_completion_cost":0.0196608,"max_cost":0.171882},"sats_pricing":{"prompt":0.0002307184727052356,"completion":0.0009228738908209423,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":5.76796181763089e-05,"input_cache_write":0.0,"max_prompt_cost":241.69513476265985,"max_completion_cost":30.24073165442064,"max_cost":264.3756835034754},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4.1-nano:batch","name":"OpenAI: GPT-4.1 Nano (batch)","created":1744651369,"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.75e-08,"completion":1.5e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":9.375e-09,"input_cache_write":0.0,"max_prompt_cost":0.039284099999999995,"max_completion_cost":0.0049152,"max_cost":0.0429705},"sats_pricing":{"prompt":5.76796181763089e-05,"completion":0.0002307184727052356,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":1.4419904544077224e-05,"input_cache_write":0.0,"max_prompt_cost":60.42378369066496,"max_completion_cost":7.56018291360516,"max_cost":66.09392087586885},"per_request_limits":null,"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","alias_ids":null,"forwarded_model_id":null},{"id":"o1-pro:batch","name":"OpenAI: o1-pro (batch)","created":1742423211,"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-05,"completion":0.000225,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":11.25,"max_completion_cost":22.5,"max_cost":28.125},"sats_pricing":{"prompt":0.08651942726446335,"completion":0.3460777090578534,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":17303.88545289267,"max_completion_cost":34607.77090578534,"max_cost":43259.71363223167},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o1-pro","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini-high:batch","name":"OpenAI: o3 Mini High (batch)","created":1739372611,"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":2.0625e-07,"input_cache_write":0.0,"max_prompt_cost":0.0825,"max_completion_cost":0.165,"max_cost":0.20625000000000002},"sats_pricing":{"prompt":0.0006344757999393979,"completion":0.0025379031997575918,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00031723789996969897,"input_cache_write":0.0,"max_prompt_cost":126.89515998787958,"max_completion_cost":253.79031997575916,"max_cost":317.23789996969896},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-mini-high-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o3-mini:batch","name":"OpenAI: o3 Mini (batch)","created":1738351721,"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":4.125e-07,"completion":1.65e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":2.0625e-07,"input_cache_write":0.0,"max_prompt_cost":0.0825,"max_completion_cost":0.165,"max_cost":0.20625000000000002},"sats_pricing":{"prompt":0.0006344757999393979,"completion":0.0025379031997575918,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.00031723789996969897,"input_cache_write":0.0,"max_prompt_cost":126.89515998787958,"max_completion_cost":253.79031997575916,"max_cost":317.23789996969896},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o3-mini-2025-01-31","alias_ids":null,"forwarded_model_id":null},{"id":"o1:batch","name":"OpenAI: o1 (batch)","created":1734459999,"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-06,"completion":2.25e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":2.8125e-06,"input_cache_write":0.0,"max_prompt_cost":1.125,"max_completion_cost":2.25,"max_cost":2.8125},"sats_pricing":{"prompt":0.008651942726446335,"completion":0.03460777090578534,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.004325971363223167,"input_cache_write":0.0,"max_prompt_cost":1730.388545289267,"max_completion_cost":3460.777090578534,"max_cost":4325.971363223168},"per_request_limits":null,"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/o1-2024-12-17","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o-mini:batch","name":"OpenAI: GPT-4o-mini (batch)","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":5.625e-08,"completion":2.25e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":2.8125e-08,"input_cache_write":0.0,"max_prompt_cost":0.0072,"max_completion_cost":0.0036864,"max_cost":0.0099648},"sats_pricing":{"prompt":8.651942726446334e-05,"completion":0.00034607770905785337,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":4.325971363223167e-05,"input_cache_write":0.0,"max_prompt_cost":11.074486689851309,"max_completion_cost":5.67013718520387,"max_cost":15.32708957875421},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o-mini","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4o:batch","name":"OpenAI: GPT-4o (batch)","created":1715558400,"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":9.375000000000001e-07,"completion":3.7500000000000005e-06,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":4.6875000000000006e-07,"input_cache_write":0.0,"max_prompt_cost":0.12000000000000002,"max_completion_cost":0.06144000000000001,"max_cost":0.16608},"sats_pricing":{"prompt":0.0014419904544077226,"completion":0.00576796181763089,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0007209952272038613,"input_cache_write":0.0,"max_prompt_cost":184.5747781641885,"max_completion_cost":94.50228642006451,"max_cost":255.45149297923686},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4o","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-4-turbo:batch","name":"OpenAI: GPT-4 Turbo (batch)","created":1712620800,"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-06,"completion":1.125e-05,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.4800000000000001,"max_completion_cost":0.04608,"max_cost":0.5107200000000001},"sats_pricing":{"prompt":0.00576796181763089,"completion":0.01730388545289267,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":738.299112656754,"max_completion_cost":70.87671481504837,"max_cost":785.5502558667863},"per_request_limits":null,"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-4-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"gpt-3.5-turbo:batch","name":"OpenAI: GPT-3.5 Turbo (batch)","created":1685232000,"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":1.875e-07,"completion":5.625e-07,"request":0.0,"image":0.0,"web_search":0.0075,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0030721875000000003,"max_completion_cost":0.002304,"max_cost":0.0046081875},"sats_pricing":{"prompt":0.0002883980908815445,"completion":0.0008651942726446336,"request":0.001,"image":0.0,"web_search":11.53592363526178,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.725402719094107,"max_completion_cost":3.543835740752419,"max_cost":7.0879598795957195},"per_request_limits":null,"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/gpt-3.5-turbo","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-multimodal-3.5","name":"VoyageAI by MongoDB: voyage-multimodal-3.5","created":1785188629,"description":"voyage-multimodal-3.5 is a state-of-the-art multimodal embedding model capable of vectorizing not only text, images, and video individually, but also content that interleaves all three modalities. It delivers excellent performance for...","context_length":32000,"architecture":{"modality":"text+image->embeddings","input_modalities":["text","image"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.999999999999999e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0028799999999999997,"max_completion_cost":0.0,"max_cost":0.0028799999999999997},"sats_pricing":{"prompt":0.00013843108362314135,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.429794675940523,"max_completion_cost":0.0,"max_cost":4.429794675940523},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-multimodal-3.5-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4-lite","name":"VoyageAI by MongoDB: voyage-4-lite","created":1785188627,"description":"voyage-4-lite is a lightweight, general-purpose embedding model optimized for low latency and cost. Enabled by Matryoshka learning and quantization-aware training, voyage-4-lite supports embeddings in 2048, 1024, 512, and 256 dimensions,...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.5000000000000002e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00048000000000000007,"max_completion_cost":0.0,"max_cost":0.00048000000000000007},"sats_pricing":{"prompt":2.3071847270523563e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.738299112656754,"max_completion_cost":0.0,"max_cost":0.738299112656754},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-lite-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4","name":"VoyageAI by MongoDB: voyage-4","created":1785188626,"description":"voyage-4 is a general-purpose (including multilingual) embedding model optimized for retrieval/search and AI applications. voyage-4 supports embeddings in 2048, 1024, 512, and 256 dimensions, with multiple quantization options. Learn more...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":4.499999999999999e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0014399999999999999,"max_completion_cost":0.0,"max_cost":0.0014399999999999999},"sats_pricing":{"prompt":6.921554181157067e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":2.2148973379702617,"max_completion_cost":0.0,"max_cost":2.2148973379702617},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"voyage-4-large","name":"VoyageAI by MongoDB: voyage-4-large","created":1785188624,"description":"voyage-4-large is a state-of-the-art general-purpose and multilingual embedding optimized for retrieval quality. Enabled by Matryoshka learning and quantization-aware training, voyage-4-large supports embeddings in 2048, 1024, 512, and 256 dimensions, with...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":8.999999999999999e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0028799999999999997,"max_completion_cost":0.0,"max_cost":0.0028799999999999997},"sats_pricing":{"prompt":0.00013843108362314135,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":4.429794675940523,"max_completion_cost":0.0,"max_cost":4.429794675940523},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"voyageai/voyage-4-large-20260727","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-2","name":"Google: Gemini Embedding 2","created":1779290135,"description":"Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.5e-07,"completion":0.0,"request":0.0,"image":3.375e-07,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0012288,"max_completion_cost":0.0,"max_cost":0.0012288},"sats_pricing":{"prompt":0.0002307184727052356,"completion":0.0,"request":0.001,"image":0.0005191165635867801,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.89004572840129,"max_completion_cost":0.0,"max_cost":1.89004572840129},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-2","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-2-preview","name":"Google: Gemini Embedding 2 Preview","created":1776436465,"description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.5e-07,"completion":0.0,"request":0.0,"image":3.375e-07,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0012288,"max_completion_cost":0.0,"max_cost":0.0012288},"sats_pricing":{"prompt":0.0002307184727052356,"completion":0.0,"request":0.001,"image":0.0005191165635867801,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.89004572840129,"max_completion_cost":0.0,"max_cost":1.89004572840129},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-2-preview","alias_ids":null,"forwarded_model_id":null},{"id":"pplx-embed-v1-4b","name":"Perplexity: Embed V1 4B","created":1773625372,"description":"pplx-embed-v1 -4B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 4B parameter model maximizing retrieval...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":2.2499999999999996e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0007199999999999999,"max_completion_cost":0.0,"max_cost":0.0007199999999999999},"sats_pricing":{"prompt":3.460777090578534e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.1074486689851308,"max_completion_cost":0.0,"max_cost":1.1074486689851308},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/pplx-embed-v1-4B","alias_ids":null,"forwarded_model_id":null},{"id":"pplx-embed-v1-0.6b","name":"Perplexity: Embed V1 0.6B","created":1773624868,"description":"pplx-embed-v1-0.6B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 0.6B parameter model targeting lightweight, low-latency...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.0000000000000004e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":9.600000000000002e-05,"max_completion_cost":0.0,"max_cost":9.600000000000002e-05},"sats_pricing":{"prompt":4.6143694541047125e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.14765982253135082,"max_completion_cost":0.0,"max_cost":0.14765982253135082},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"perplexity/pplx-embed-v1-0.6B","alias_ids":null,"forwarded_model_id":null},{"id":"gte-base","name":"Thenlper: GTE-Base","created":1763433820,"description":"The gte-base embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, delivering efficient and effective semantic embeddings optimized for textual similarity, semantic search, and clustering applications.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thenlper/gte-base-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"gte-large","name":"Thenlper: GTE-Large","created":1763433655,"description":"The gte-large embedding model converts English sentences, paragraphs and moderate-length documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for information retrieval, semantic textual similarity, reranking and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.500000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.8400000000000005e-06,"max_completion_cost":0.0,"max_cost":3.8400000000000005e-06},"sats_pricing":{"prompt":1.1535923635261782e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005906392901254032,"max_completion_cost":0.0,"max_cost":0.005906392901254032},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"thenlper/gte-large-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"e5-large-v2","name":"Intfloat: E5-Large-v2","created":1763433432,"description":"The e5-large-v2 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-accuracy semantic embeddings optimized for retrieval, semantic search, reranking, and similarity-scoring tasks.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.500000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.8400000000000005e-06,"max_completion_cost":0.0,"max_cost":3.8400000000000005e-06},"sats_pricing":{"prompt":1.1535923635261782e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005906392901254032,"max_completion_cost":0.0,"max_cost":0.005906392901254032},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/e5-large-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"e5-base-v2","name":"Intfloat: E5-Base-v2","created":1763433192,"description":"The e5-base-v2 embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, similarity scoring,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/e5-base-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"multilingual-e5-large","name":"Intfloat: Multilingual-E5-Large","created":1763433047,"description":"The multilingual-e5-large embedding model encodes sentences, paragraphs, and documents across over 90 languages into a 1024-dimensional dense vector space, delivering robust semantic embeddings optimized for multilingual retrieval, cross-language similarity, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.500000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.8400000000000005e-06,"max_completion_cost":0.0,"max_cost":3.8400000000000005e-06},"sats_pricing":{"prompt":1.1535923635261782e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005906392901254032,"max_completion_cost":0.0,"max_cost":0.005906392901254032},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"intfloat/multilingual-e5-large-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"paraphrase-minilm-l6-v2","name":"Sentence Transformers: paraphrase-MiniLM-L6-v2","created":1763432454,"description":"The paraphrase-MiniLM-L6-v2 embedding model converts sentences and short paragraphs into a 384-dimensional dense vector space, producing high-quality semantic embeddings optimized for paraphrase detection, semantic similarity scoring, clustering, and lightweight retrieval...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/paraphrase-minilm-l6-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-minilm-l12-v2","name":"Sentence Transformers: all-MiniLM-L12-v2","created":1763432155,"description":"The all-MiniLM-L12-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, clustering, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-minilm-l12-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-base-en-v1.5","name":"BAAI: bge-base-en-v1.5","created":1763431837,"description":"The bge-base-en-v1.5 embedding model converts English sentences and paragraphs into 768-dimensional dense vectors, delivering efficient, high-quality semantic embeddings optimized for retrieval, semantic search, and document-matching workflows. This version (v1.5) features...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-base-en-v1.5-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"multi-qa-mpnet-base-dot-v1","name":"Sentence Transformers: multi-qa-mpnet-base-dot-v1","created":1763431339,"description":"The multi-qa-mpnet-base-dot-v1 embedding model transforms sentences and short paragraphs into a 768-dimensional dense vector space, generating high-quality semantic embeddings optimized for question-and-answer retrieval, semantic search, and similarity-scoring across diverse content.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-large-en-v1.5","name":"BAAI: bge-large-en-v1.5","created":1763431087,"description":"The bge-large-en-v1.5 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-fidelity semantic embeddings optimized for semantic search, document retrieval, and downstream NLP tasks...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.500000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.8400000000000005e-06,"max_completion_cost":0.0,"max_cost":3.8400000000000005e-06},"sats_pricing":{"prompt":1.1535923635261782e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.005906392901254032,"max_completion_cost":0.0,"max_cost":0.005906392901254032},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-large-en-v1.5-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"bge-m3","name":"BAAI: bge-m3","created":1763424372,"description":"The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.","context_length":8194,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.500000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":6.144000000000001e-05,"max_completion_cost":0.0,"max_cost":6.144000000000001e-05},"sats_pricing":{"prompt":1.1535923635261782e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.09450228642006452,"max_completion_cost":0.0,"max_cost":0.09450228642006452},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"baai/bge-m3-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-mpnet-base-v2","name":"Sentence Transformers: all-mpnet-base-v2","created":1763421830,"description":"The all-mpnet-base-v2 embedding model encodes sentences and short paragraphs into a 768-dimensional dense vector space, providing high-fidelity semantic embeddings well suited for tasks like information retrieval, clustering, similarity scoring, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-mpnet-base-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"all-minilm-l6-v2","name":"Sentence Transformers: all-MiniLM-L6-v2","created":1763421176,"description":"The all-MiniLM-L6-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, enabling high-quality semantic representations that are ideal for downstream tasks such as information retrieval, clustering,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":3.7500000000000005e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.9200000000000003e-06,"max_completion_cost":0.0,"max_cost":1.9200000000000003e-06},"sats_pricing":{"prompt":5.767961817630891e-06,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.002953196450627016,"max_completion_cost":0.0,"max_cost":0.002953196450627016},"per_request_limits":null,"top_provider":{"context_length":512,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"sentence-transformers/all-minilm-l6-v2-20251117","alias_ids":null,"forwarded_model_id":null},{"id":"mistral-embed-2312","name":"Mistral: Mistral Embed 2312","created":1761944622,"description":"Mistral Embed is a specialized embedding model for text data, optimized for semantic search and RAG applications. Developed by Mistral AI in late 2023, it produces 1024-dimensional vectors that effectively...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":7.5e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0006144,"max_completion_cost":0.0,"max_cost":0.0006144},"sats_pricing":{"prompt":0.0001153592363526178,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.945022864200645,"max_completion_cost":0.0,"max_cost":0.945022864200645},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/mistral-embed-2312","alias_ids":null,"forwarded_model_id":null},{"id":"gemini-embedding-001","name":"Google: Gemini Embedding 001","created":1761943410,"description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","context_length":20000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00225,"max_completion_cost":0.0,"max_cost":0.00225},"sats_pricing":{"prompt":0.00017303885452892668,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":3.4607770905785338,"max_completion_cost":0.0,"max_cost":3.4607770905785338},"per_request_limits":null,"top_provider":{"context_length":20000,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"google/gemini-embedding-001","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-ada-002","name":"OpenAI: Text Embedding Ada 002","created":1761865798,"description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.5e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0006144,"max_completion_cost":0.0,"max_cost":0.0006144},"sats_pricing":{"prompt":0.0001153592363526178,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.945022864200645,"max_completion_cost":0.0,"max_cost":0.945022864200645},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/text-embedding-ada-002","alias_ids":["text-embedding-ada-002-v2"],"forwarded_model_id":null},{"id":"codestral-embed-2505","name":"Mistral: Codestral Embed 2505","created":1761864460,"description":"Mistral Codestral Embed is specially designed for code, perfect for embedding code databases, repositories, and powering coding assistants with state-of-the-art retrieval.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":1.125e-07,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0009216,"max_completion_cost":0.0,"max_cost":0.0009216},"sats_pricing":{"prompt":0.00017303885452892668,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.4175342963009674,"max_completion_cost":0.0,"max_cost":1.4175342963009674},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"mistralai/codestral-embed-2505","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-3-large","name":"OpenAI: Text Embedding 3 Large","created":1761862866,"description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":9.749999999999999e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0007987199999999999,"max_completion_cost":0.0,"max_cost":0.0007987199999999999},"sats_pricing":{"prompt":0.00014996700725840312,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":1.2285297234608383,"max_completion_cost":0.0,"max_cost":1.2285297234608383},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/text-embedding-3-large","alias_ids":null,"forwarded_model_id":null},{"id":"text-embedding-3-small","name":"OpenAI: Text Embedding 3 Small","created":1761857455,"description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.5000000000000002e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00012288000000000002,"max_completion_cost":0.0,"max_cost":0.00012288000000000002},"sats_pricing":{"prompt":2.3071847270523563e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.18900457284012903,"max_completion_cost":0.0,"max_cost":0.18900457284012903},"per_request_limits":null,"top_provider":{"context_length":8192,"max_completion_tokens":null,"is_moderated":true},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"openai/text-embedding-3-small","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-embedding-8b","name":"Qwen: Qwen3 Embedding 8B","created":1761680622,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":7.500000000000001e-09,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.00024000000000000003,"max_completion_cost":0.0,"max_cost":0.00024000000000000003},"sats_pricing":{"prompt":1.1535923635261782e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.369149556328377,"max_completion_cost":0.0,"max_cost":0.369149556328377},"per_request_limits":null,"top_provider":{"context_length":32000,"max_completion_tokens":32000,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-embedding-8b","alias_ids":null,"forwarded_model_id":null},{"id":"qwen3-embedding-4b","name":"Qwen: Qwen3 Embedding 4B","created":1761662922,"description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":1.5000000000000002e-08,"completion":0.0,"request":0.0,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.0004915200000000001,"max_completion_cost":0.0,"max_cost":0.0004915200000000001},"sats_pricing":{"prompt":2.3071847270523563e-05,"completion":0.0,"request":0.001,"image":0.0,"web_search":0.0,"internal_reasoning":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"max_prompt_cost":0.7560182913605161,"max_completion_cost":0.0,"max_cost":0.7560182913605161},"per_request_limits":null,"top_provider":{"context_length":32768,"max_completion_tokens":null,"is_moderated":false},"enabled":true,"upstream_provider_id":"openrouter","canonical_slug":"qwen/qwen3-embedding-4b","alias_ids":null,"forwarded_model_id":null}]}