{"data":[{"id":"deepseek/deepseek-v4-flash-0731","canonical_slug":"deepseek/deepseek-v4-flash-20260731","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","created":1785478908,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","context_length":1310720,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000004","completion":"0.00000008","input_cache_read":"0.000000008"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":[],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260731/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1237,"win_rate":48.6,"rank":37},{"arena":"models","category":"asciiart","elo":1110,"win_rate":33.3,"rank":58},{"arena":"models","category":"codecategories","elo":1245,"win_rate":46.5,"rank":40},{"arena":"models","category":"dataviz","elo":1197,"win_rate":41.2,"rank":55},{"arena":"models","category":"gamedev","elo":1233,"win_rate":44.7,"rank":38},{"arena":"models","category":"svg","elo":1213,"win_rate":45.1,"rank":25},{"arena":"models","category":"uicomponent","elo":1252,"win_rate":47,"rank":35},{"arena":"models","category":"website","elo":1251,"win_rate":46.8,"rank":38}],"artificial_analysis":{"intelligence_index":34.5,"coding_index":69.1,"agentic_index":41.7}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"high"}},{"id":"openai/gpt-5.6-luna","canonical_slug":"openai/gpt-5.6-luna-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000012","web_search":"0.01","input_cache_read":"0.00000002","input_cache_write":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000004","completion":"0.0000018","input_cache_read":"0.00000004","input_cache_write":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-luna-20260709/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":37.5,"coding_index":71.4,"agentic_index":42.7}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"deepseek/deepseek-v4-flash","canonical_slug":"deepseek/deepseek-v4-flash-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek: DeepSeek V4 Flash 0423","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000006594","completion":"0.00000013188","input_cache_read":"0.000000013188"},"top_provider":{"context_length":1024000,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260423/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1217,"win_rate":49.3,"rank":44},{"arena":"models","category":"asciiart","elo":1128,"win_rate":42.8,"rank":53},{"arena":"models","category":"codecategories","elo":1220,"win_rate":48.9,"rank":47},{"arena":"models","category":"dataviz","elo":1143,"win_rate":40.7,"rank":79},{"arena":"models","category":"gamedev","elo":1220,"win_rate":50.2,"rank":43},{"arena":"models","category":"svg","elo":1181,"win_rate":48.4,"rank":36},{"arena":"models","category":"uicomponent","elo":1179,"win_rate":44.7,"rank":62},{"arena":"models","category":"website","elo":1220,"win_rate":49.1,"rank":48}],"artificial_analysis":{"intelligence_index":24.8,"coding_index":52,"agentic_index":27.9}},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"],"default_effort":"high"}},{"id":"qwen/qwen3.7-flash","canonical_slug":"qwen/qwen3.7-flash-20260727","hugging_face_id":null,"name":"Qwen: Qwen3.7 Flash","created":1785190561,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000013","input_cache_read":"0.000000006","input_cache_write":"0.000000038","overrides":[{"min_prompt_tokens":32000,"prompt":"0.0000001","completion":"0.0000004","input_cache_read":"0.00000002","input_cache_write":"0.000000125"},{"min_prompt_tokens":256000,"prompt":"0.0000002","completion":"0.0000008","input_cache_read":"0.00000004","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.7-flash-20260727/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"z-ai/glm-5.3-flash","canonical_slug":"z-ai/glm-5.3-flash-20260826","hugging_face_id":"zai-org/GLM-5.3-Flash","name":"Z.ai: GLM 5.3 Flash","created":1787752741,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","context_length":1310720,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2098-12-31","links":{"details":"/api/v1/models/z-ai/glm-5.3-flash-20260826/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1366,"win_rate":63,"rank":5},{"arena":"models","category":"asciiart","elo":1297,"win_rate":57.7,"rank":7},{"arena":"models","category":"codecategories","elo":1303,"win_rate":51.5,"rank":15},{"arena":"models","category":"dataviz","elo":1273,"win_rate":50.7,"rank":22},{"arena":"models","category":"gamedev","elo":1322,"win_rate":50.5,"rank":11},{"arena":"models","category":"svg","elo":1314,"win_rate":57.9,"rank":7},{"arena":"models","category":"uicomponent","elo":1336,"win_rate":56.8,"rank":7},{"arena":"models","category":"website","elo":1290,"win_rate":49.9,"rank":18}],"artificial_analysis":{"intelligence_index":41.9,"coding_index":71.5,"agentic_index":51.2}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"tencent/hy4-preview","canonical_slug":"tencent/hy4-preview-20260827","hugging_face_id":"tencent/Hy4-preview","name":"Tencent: Hy4 preview","created":1787897375,"description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},"top_provider":{"context_length":1048576,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/tencent/hy4-preview-20260827/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["high","low","none"],"default_effort":"high"}},{"id":"openai/gpt-oss-120b","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000037","completion":"0.00000017"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":930,"win_rate":29.4,"rank":109},{"arena":"models","category":"codecategories","elo":978,"win_rate":33.4,"rank":117},{"arena":"models","category":"dataviz","elo":1003,"win_rate":43.6,"rank":107},{"arena":"models","category":"gamedev","elo":1015,"win_rate":40.5,"rank":106},{"arena":"models","category":"uicomponent","elo":941,"win_rate":35.7,"rank":112},{"arena":"models","category":"website","elo":980,"win_rate":32.5,"rank":118}],"artificial_analysis":{"intelligence_index":12.3,"coding_index":30.4,"agentic_index":6.2}},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-5.6-sol","canonical_slug":"openai/gpt-5.6-sol-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-sol-20260709/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":47.1,"coding_index":77.4,"agentic_index":50.5}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"z-ai/glm-5.2","canonical_slug":"z-ai/glm-5.2-20260616","hugging_face_id":"zai-org/GLM-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000966","completion":"0.000003036","input_cache_read":"0.0000001932"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.2-20260616/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1205,"win_rate":49.1,"rank":10},{"arena":"agents","category":"androidnative","elo":1193,"win_rate":53,"rank":17},{"arena":"agents","category":"fullstack","elo":1250,"win_rate":61.2,"rank":12},{"arena":"agents","category":"godotgamedev","elo":1142,"win_rate":40.1,"rank":16},{"arena":"agents","category":"htmlslides","elo":1178,"win_rate":49.6,"rank":13},{"arena":"agents","category":"mobileapps","elo":1193,"win_rate":51.4,"rank":18},{"arena":"agents","category":"python-pptxslides","elo":1194,"win_rate":48.7,"rank":16},{"arena":"agents","category":"webapps","elo":1239,"win_rate":56.3,"rank":12},{"arena":"models","category":"3d","elo":1325,"win_rate":55.6,"rank":12},{"arena":"models","category":"asciiart","elo":1229,"win_rate":50.1,"rank":18},{"arena":"models","category":"codecategories","elo":1311,"win_rate":54.5,"rank":10},{"arena":"models","category":"dataviz","elo":1315,"win_rate":54.3,"rank":9},{"arena":"models","category":"gamedev","elo":1302,"win_rate":51.9,"rank":15},{"arena":"models","category":"svg","elo":1245,"win_rate":52.1,"rank":15},{"arena":"models","category":"uicomponent","elo":1309,"win_rate":55.7,"rank":13},{"arena":"models","category":"website","elo":1306,"win_rate":54.7,"rank":11}],"artificial_analysis":{"intelligence_index":null,"coding_index":68.8,"agentic_index":39.4}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["xhigh","high"],"default_effort":"high"}},{"id":"anthropic/claude-sonnet-4.6","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-sonnet-20260217/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1177,"win_rate":52.2,"rank":16},{"arena":"agents","category":"androidnative","elo":1214,"win_rate":62,"rank":13},{"arena":"agents","category":"fullstack","elo":1223,"win_rate":64.3,"rank":15},{"arena":"agents","category":"godotgamedev","elo":1226,"win_rate":60.6,"rank":6},{"arena":"agents","category":"mobileapps","elo":1243,"win_rate":63.6,"rank":10},{"arena":"agents","category":"webapps","elo":1209,"win_rate":56.9,"rank":20},{"arena":"models","category":"3d","elo":1263,"win_rate":57.6,"rank":30},{"arena":"models","category":"asciiart","elo":1251,"win_rate":60.1,"rank":16},{"arena":"models","category":"codecategories","elo":1291,"win_rate":59.4,"rank":18},{"arena":"models","category":"dataviz","elo":1295,"win_rate":58.2,"rank":13},{"arena":"models","category":"gamedev","elo":1282,"win_rate":58.8,"rank":25},{"arena":"models","category":"svg","elo":1222,"win_rate":58.7,"rank":20},{"arena":"models","category":"uicomponent","elo":1286,"win_rate":58.3,"rank":21},{"arena":"models","category":"website","elo":1297,"win_rate":60,"rank":14}],"artificial_analysis":{"intelligence_index":30.5,"coding_index":63,"agentic_index":33.1}},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"],"default_effort":"medium"}},{"id":"google/gemini-2.5-flash","canonical_slug":"google/gemini-2.5-flash","hugging_face_id":"","name":"Google: Gemini 2.5 Flash","created":1750172488,"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-flash/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1100,"win_rate":47.4,"rank":89},{"arena":"models","category":"codecategories","elo":1119,"win_rate":46.9,"rank":90},{"arena":"models","category":"dataviz","elo":1150,"win_rate":49.1,"rank":77},{"arena":"models","category":"gamedev","elo":1088,"win_rate":44.3,"rank":93},{"arena":"models","category":"uicomponent","elo":1107,"win_rate":48.9,"rank":86},{"arena":"models","category":"website","elo":1127,"win_rate":47.1,"rank":90},{"arena":"models","category":"svg","elo":1045,"win_rate":43.1,"rank":74}]},"reasoning":{"mandatory":false}},{"id":"google/gemini-2.5-flash-lite","canonical_slug":"google/gemini-2.5-flash-lite","hugging_face_id":"","name":"Google: Gemini 2.5 Flash Lite","created":1753200276,"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","image":"0.0000001","audio":"0.0000003","input_audio_cache":"0.00000003","web_search":"0.014","internal_reasoning":"0.0000004","input_cache_read":"0.00000001","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-2.5-flash-lite/endpoints"},"reasoning":{"mandatory":false}},{"id":"amazon/nova-micro-v1","canonical_slug":"amazon/nova-micro-v1","hugging_face_id":"","name":"Amazon: Nova Micro 1.0","created":1733437237,"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.000000035","completion":"0.00000014"},"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/api/v1/models/amazon/nova-micro-v1/endpoints"}},{"id":"anthropic/claude-sonnet-5","canonical_slug":"anthropic/claude-sonnet-5-20260630","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-sonnet-5-20260630/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1226,"win_rate":52.3,"rank":7},{"arena":"agents","category":"androidnative","elo":1246,"win_rate":54.8,"rank":9},{"arena":"agents","category":"fullstack","elo":1263,"win_rate":55.6,"rank":9},{"arena":"agents","category":"godotgamedev","elo":1270,"win_rate":60.6,"rank":2},{"arena":"agents","category":"htmlslides","elo":1220,"win_rate":53.7,"rank":5},{"arena":"agents","category":"mobileapps","elo":1260,"win_rate":56.3,"rank":5},{"arena":"agents","category":"python-pptxslides","elo":1216,"win_rate":50.1,"rank":13},{"arena":"agents","category":"webapps","elo":1263,"win_rate":55,"rank":8},{"arena":"models","category":"3d","elo":1288,"win_rate":54.8,"rank":21},{"arena":"models","category":"asciiart","elo":1225,"win_rate":52.1,"rank":19},{"arena":"models","category":"codecategories","elo":1292,"win_rate":53.8,"rank":17},{"arena":"models","category":"dataviz","elo":1260,"win_rate":52.8,"rank":26},{"arena":"models","category":"gamedev","elo":1312,"win_rate":54,"rank":14},{"arena":"models","category":"svg","elo":1215,"win_rate":51.5,"rank":23},{"arena":"models","category":"uicomponent","elo":1296,"win_rate":55.1,"rank":17},{"arena":"models","category":"website","elo":1289,"win_rate":53.6,"rank":19}],"artificial_analysis":{"intelligence_index":38.4,"coding_index":71.5,"agentic_index":44.3}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"google/gemini-3.7-flash","canonical_slug":"google/gemini-3.7-flash-20260813","hugging_face_id":null,"name":"Google: Gemini 3.7 Flash","created":1786640581,"description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.7-flash-20260813/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1235,"win_rate":52.1,"rank":5},{"arena":"agents","category":"androidnative","elo":1259,"win_rate":53.6,"rank":5},{"arena":"agents","category":"fullstack","elo":1222,"win_rate":45.6,"rank":16},{"arena":"agents","category":"mobileapps","elo":1260,"win_rate":53.9,"rank":6},{"arena":"agents","category":"webapps","elo":1241,"win_rate":48.4,"rank":11},{"arena":"models","category":"3d","elo":1338,"win_rate":59.1,"rank":9},{"arena":"models","category":"asciiart","elo":1259,"win_rate":52,"rank":15},{"arena":"models","category":"codecategories","elo":1320,"win_rate":57,"rank":9},{"arena":"models","category":"dataviz","elo":1329,"win_rate":58.5,"rank":6},{"arena":"models","category":"gamedev","elo":1342,"win_rate":56.9,"rank":7},{"arena":"models","category":"uicomponent","elo":1306,"win_rate":52.7,"rank":15},{"arena":"models","category":"website","elo":1314,"win_rate":56.9,"rank":7}],"artificial_analysis":{"intelligence_index":39.4,"coding_index":76.1,"agentic_index":36.4}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"google/gemini-3.8-flash","canonical_slug":"google/gemini-3.8-flash-20260902","hugging_face_id":null,"name":"Google: Gemini 3.8 Flash","created":1788362056,"description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.8-flash-20260902/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"fullstack","elo":1284,"win_rate":48.5,"rank":6},{"arena":"agents","category":"python-pptxslides","elo":1175,"win_rate":39.4,"rank":19},{"arena":"models","category":"3d","elo":1323,"win_rate":54.5,"rank":13},{"arena":"models","category":"codecategories","elo":1322,"win_rate":55,"rank":8},{"arena":"models","category":"dataviz","elo":1257,"win_rate":49,"rank":28},{"arena":"models","category":"gamedev","elo":1338,"win_rate":56.3,"rank":8},{"arena":"models","category":"uicomponent","elo":1343,"win_rate":56.9,"rank":5},{"arena":"models","category":"website","elo":1314,"win_rate":54.6,"rank":8}],"artificial_analysis":{"intelligence_index":41.2,"coding_index":76.3,"agentic_index":41.1}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-4o-mini","canonical_slug":"openai/gpt-4o-mini","hugging_face_id":null,"name":"OpenAI: GPT-4o-mini","created":1721260800,"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-4o-mini/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":null,"coding_index":11.4,"agentic_index":null}}},{"id":"google/gemma-4-31b-it","canonical_slug":"google/gemma-4-31b-it-20260402","hugging_face_id":"google/gemma-4-31B-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-31b-it-20260402/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":15.4,"coding_index":43.4,"agentic_index":6.7}},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"anthropic/claude-opus-5","canonical_slug":"anthropic/claude-opus-5-20260723","hugging_face_id":null,"name":"Claude Opus 5","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-5-20260723/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1262,"win_rate":53.8,"rank":2},{"arena":"agents","category":"androidnative","elo":1267,"win_rate":54.8,"rank":3},{"arena":"agents","category":"fullstack","elo":1330,"win_rate":62.3,"rank":3},{"arena":"agents","category":"mobileapps","elo":1348,"win_rate":67.1,"rank":1},{"arena":"agents","category":"python-pptxslides","elo":1264,"win_rate":54,"rank":5},{"arena":"agents","category":"webapps","elo":1280,"win_rate":56.6,"rank":4},{"arena":"models","category":"3d","elo":1364,"win_rate":61.7,"rank":6},{"arena":"models","category":"asciiart","elo":1389,"win_rate":70.8,"rank":1},{"arena":"models","category":"codecategories","elo":1338,"win_rate":58.3,"rank":4},{"arena":"models","category":"dataviz","elo":1355,"win_rate":61.1,"rank":4},{"arena":"models","category":"gamedev","elo":1363,"win_rate":59.4,"rank":5},{"arena":"models","category":"svg","elo":1350,"win_rate":62.1,"rank":2},{"arena":"models","category":"uicomponent","elo":1360,"win_rate":61.4,"rank":3},{"arena":"models","category":"website","elo":1320,"win_rate":56.4,"rank":5}],"artificial_analysis":{"intelligence_index":50.7,"coding_index":78,"agentic_index":56.2}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"google/gemma-4-26b-a4b-it","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","hugging_face_id":"google/gemma-4-26B-A4B-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.000000042","completion":"0.00000022"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":null,"coding_index":39.3,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":false}}],"total_count":20,"links":{"next":null}}