{"data":[{"id":"anthropic/claude-haiku-5.5","canonical_slug":"anthropic/claude-haiku-5.5-20261007","hugging_face_id":null,"name":"Anthropic: Claude Haiku 5.5","created":1791397883,"description":"Claude Haiku 5.5 is Anthropic's small, fast model for high-volume, cost-sensitive work such as summarization, subagents, and browser use. It succeeds Claude Haiku 4.5 with stronger coding, computer use, and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000005","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","input_cache_write_1h":"0.0000002","overrides":[{"min_prompt_tokens":100000,"prompt":"0.0000005","completion":"0.0000025","input_cache_read":"0.00000005","input_cache_write":"0.000000625","input_cache_write_1h":"0.000001"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-haiku-5.5-20261007/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"gamedev","elo":1361,"win_rate":59.7,"rank":5},{"arena":"models","category":"website","elo":1269,"win_rate":48.3,"rank":31}],"artificial_analysis":{"intelligence_index":43.4,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-6.1-sol-pro","canonical_slug":"openai/gpt-6.1-sol-pro-20260929","hugging_face_id":null,"name":"OpenAI: GPT-6.1 Sol Pro","created":1790702886,"description":"GPT-6.1 Sol Pro is the same underlying model as [GPT-6.1 Sol](https://openrouter.ai/openai/gpt-6.1-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000002","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6.1-sol-pro-20260929/endpoints"},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-6.1-sol","canonical_slug":"openai/gpt-6.1-sol-20260929","hugging_face_id":null,"name":"OpenAI: GPT-6.1 Sol","created":1790702882,"description":"GPT-6.1 Sol is an upgrade to GPT-6 Sol from OpenAI, positioned below the flagship GPT-6 Astra in the GPT-6 series. It is suited for agentic coding, computer use, document-heavy professional...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000002","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6.1-sol-20260929/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"codecategories","elo":1345,"win_rate":58.1,"rank":5},{"arena":"models","category":"gamedev","elo":1354,"win_rate":58.8,"rank":6},{"arena":"models","category":"website","elo":1322,"win_rate":55.6,"rank":4}],"artificial_analysis":{"intelligence_index":51.8,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"medium"}},{"id":"anthropic/claude-sonnet-5.5","canonical_slug":"anthropic/claude-sonnet-5.5-20260928","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5.5","created":1790618686,"description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-sonnet-5.5-20260928/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"codecategories","elo":1358,"win_rate":59.2,"rank":4},{"arena":"models","category":"dataviz","elo":1383,"win_rate":66.2,"rank":2},{"arena":"models","category":"gamedev","elo":1442,"win_rate":68.3,"rank":2},{"arena":"models","category":"uicomponent","elo":1382,"win_rate":64.3,"rank":2},{"arena":"models","category":"website","elo":1307,"win_rate":52.3,"rank":10}],"artificial_analysis":{"intelligence_index":56,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"openai/gpt-6-luna-pro","canonical_slug":"openai/gpt-6-luna-pro-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Luna Pro","created":1790100791,"description":"GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000005","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.00000075","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6-luna-pro-20260922/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"openai/gpt-6-luna","canonical_slug":"openai/gpt-6-luna-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Luna","created":1790100786,"description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000005","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.00000075","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6-luna-20260922/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":38.1,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"openai/gpt-6-sol-pro","canonical_slug":"openai/gpt-6-sol-pro-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Sol Pro","created":1790100781,"description":"GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6-sol-pro-20260922/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"openai/gpt-6-sol","canonical_slug":"openai/gpt-6-sol-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Sol","created":1790100775,"description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6-sol-20260922/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":47.6,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"anthropic/claude-opus-5.5","canonical_slug":"anthropic/claude-opus-5.5-20260921","hugging_face_id":null,"name":"Anthropic: Claude Opus 5.5","created":1790094732,"description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-5.5-20260921/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"mobileapps","elo":1328,"win_rate":64.3,"rank":2},{"arena":"models","category":"3d","elo":1484,"win_rate":72.2,"rank":1},{"arena":"models","category":"codecategories","elo":1403,"win_rate":64.8,"rank":1},{"arena":"models","category":"dataviz","elo":1403,"win_rate":64.7,"rank":1},{"arena":"models","category":"gamedev","elo":1462,"win_rate":72.2,"rank":1},{"arena":"models","category":"svg","elo":1331,"win_rate":50.5,"rank":1},{"arena":"models","category":"uicomponent","elo":1396,"win_rate":65.4,"rank":1},{"arena":"models","category":"website","elo":1335,"win_rate":56.9,"rank":3}],"artificial_analysis":{"intelligence_index":57.6,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"xiaomi/mimo-v2.6-flash","canonical_slug":"xiaomi/mimo-v2.6-flash-20260921","hugging_face_id":"XiaomiMiMo/MiMo-V2.6-Flash-RL","name":"Xiaomi: MiMo-V2.6-Flash","created":1790021264,"description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":0},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2.6-flash-20260921/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":37.9,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"xiaomi/mimo-v2.6-pro","canonical_slug":"xiaomi/mimo-v2.6-pro-20260921","hugging_face_id":"XiaomiMiMo/MiMo-V2.6-Pro-RL","name":"Xiaomi: MiMo-V2.6-Pro","created":1790021259,"description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":0},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2.6-pro-20260921/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"webapps","elo":1283,"win_rate":58.3,"rank":5},{"arena":"models","category":"3d","elo":1337,"win_rate":56.9,"rank":8},{"arena":"models","category":"codecategories","elo":1321,"win_rate":52.4,"rank":8},{"arena":"models","category":"dataviz","elo":1340,"win_rate":58.7,"rank":8},{"arena":"models","category":"gamedev","elo":1319,"win_rate":52.8,"rank":13},{"arena":"models","category":"uicomponent","elo":1340,"win_rate":59,"rank":8},{"arena":"models","category":"website","elo":1313,"win_rate":50.6,"rank":8}],"artificial_analysis":{"intelligence_index":46.3,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"x-ai/grok-4.7","canonical_slug":"x-ai/grok-4.7-20260916","hugging_face_id":null,"name":"SpaceXAI: Grok 4.7","created":1790007541,"description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-4.7-20260916/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1280,"win_rate":48.9,"rank":18},{"arena":"models","category":"codecategories","elo":1270,"win_rate":47.6,"rank":28},{"arena":"models","category":"dataviz","elo":1239,"win_rate":44.4,"rank":36},{"arena":"models","category":"gamedev","elo":1310,"win_rate":53.1,"rank":15},{"arena":"models","category":"uicomponent","elo":1287,"win_rate":50.2,"rank":22},{"arena":"models","category":"website","elo":1252,"win_rate":45.4,"rank":40}],"artificial_analysis":{"intelligence_index":46.4,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["xhigh","high","medium","low"],"default_effort":"high"}},{"id":"inclusionai/ling-3.0-flash-vl","canonical_slug":"inclusionai/ling-3.0-flash-vl-20260910","hugging_face_id":"inclusionAI/Ling-3.0-flash-VL","name":"inclusionAI: Ling 3.0 Flash VL","created":1789056114,"description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000021","completion":"0.0000000616","input_cache_read":"0.0000000042"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/inclusionai/ling-3.0-flash-vl-20260910/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":24.6,"coding_index":57,"agentic_index":28.7}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"deepseek/deepseek-v4.1-flash","canonical_slug":"deepseek/deepseek-v4.1-flash-20260910","hugging_face_id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek: DeepSeek V4.1 Flash","created":1789021285,"description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.000000006"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4.1-flash-20260910/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":39.5,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"high"}},{"id":"deepseek/deepseek-v4.1-flash:batch","canonical_slug":"deepseek/deepseek-v4.1-flash-20260910","hugging_face_id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek: DeepSeek V4.1 Flash (batch)","created":1789021285,"description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000112","completion":"0.000000336","input_cache_read":"0.00000000336"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4.1-flash-20260910/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":39.5,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"high"}},{"id":"openai/gpt-6-astra","canonical_slug":"openai/gpt-6-astra-20260903","hugging_face_id":null,"name":"OpenAI: GPT-6 Astra","created":1788552838,"description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00002","completion":"0.000075","input_cache_read":"0.000002","input_cache_write":"0.000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6-astra-20260903/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":52.7,"coding_index":76.9,"agentic_index":51}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-6-astra-pro","canonical_slug":"openai/gpt-6-astra-pro-20260903","hugging_face_id":null,"name":"OpenAI: GPT-6 Astra Pro","created":1788552835,"description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00002","completion":"0.000075","input_cache_read":"0.000002","input_cache_write":"0.000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-6-astra-pro-20260903/endpoints"},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"medium"}},{"id":"qwen/qwen3.8-max-0902","canonical_slug":"qwen/qwen3.8-max-20260902","hugging_face_id":null,"name":"Qwen: Qwen3.8 Max (0902)","created":1788469704,"description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.8-max-20260902/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":45.4,"coding_index":76.2,"agentic_index":56}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["xhigh","high","medium","low","minimal"],"default_effort":"xhigh"}},{"id":"ibm-granite/granite-4.2-8b","canonical_slug":"ibm-granite/granite-4.2-8b-20260831","hugging_face_id":"ibm-granite/granite-4.2-8b","name":"IBM: Granite 4.2 8B","created":1788206780,"description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000025","input_cache_read":"0.000000015"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/ibm-granite/granite-4.2-8b-20260831/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":11.1,"coding_index":22.4,"agentic_index":1.3}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["high","low","none"],"default_effort":"high"}},{"id":"qwen/qwen3.8-flash","canonical_slug":"qwen/qwen3.8-flash-20260826","hugging_face_id":"Qwen/Qwen3.8-Flash-Next","name":"Qwen: Qwen3.8 Flash","created":1787773060,"description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.8-flash-20260826/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"z-ai/glm-5.3-flash","canonical_slug":"z-ai/glm-5.3-flash-20260826","hugging_face_id":"zai-org/GLM-5.3-Flash","name":"Z.ai: GLM 5.3 Flash","created":1787752741,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003"},"top_provider":{"context_length":1048575,"max_completion_tokens":943717,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.3-flash-20260826/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1323,"win_rate":56.5,"rank":11},{"arena":"models","category":"asciiart","elo":1272,"win_rate":55,"rank":12},{"arena":"models","category":"codecategories","elo":1287,"win_rate":49.4,"rank":19},{"arena":"models","category":"dataviz","elo":1266,"win_rate":50.3,"rank":27},{"arena":"models","category":"gamedev","elo":1288,"win_rate":47.2,"rank":19},{"arena":"models","category":"svg","elo":1288,"win_rate":57.1,"rank":10},{"arena":"models","category":"uicomponent","elo":1324,"win_rate":55.1,"rank":9},{"arena":"models","category":"website","elo":1276,"win_rate":48.4,"rank":24}],"artificial_analysis":{"intelligence_index":41.8,"coding_index":71.5,"agentic_index":50.9}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"z-ai/glm-5.3-flash:batch","canonical_slug":"z-ai/glm-5.3-flash-20260826","hugging_face_id":"zai-org/GLM-5.3-Flash","name":"Z.ai: GLM 5.3 Flash (batch)","created":1787752741,"description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.0000002","input_cache_read":"0.000000012"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.3-flash-20260826/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1323,"win_rate":56.5,"rank":11},{"arena":"models","category":"asciiart","elo":1272,"win_rate":55,"rank":12},{"arena":"models","category":"codecategories","elo":1287,"win_rate":49.4,"rank":19},{"arena":"models","category":"dataviz","elo":1266,"win_rate":50.3,"rank":27},{"arena":"models","category":"gamedev","elo":1288,"win_rate":47.2,"rank":19},{"arena":"models","category":"svg","elo":1288,"win_rate":57.1,"rank":10},{"arena":"models","category":"uicomponent","elo":1324,"win_rate":55.1,"rank":9},{"arena":"models","category":"website","elo":1276,"win_rate":48.4,"rank":24}],"artificial_analysis":{"intelligence_index":41.8,"coding_index":71.5,"agentic_index":50.9}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"deepseek/deepseek-v4-flash-vision-exp","canonical_slug":"deepseek/deepseek-v4-flash-vision-exp-20260821","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek: DeepSeek V4 Flash Vision Exp","created":1787311563,"description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"top_provider":{"context_length":1048576,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-vision-exp-20260821/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":34.8,"coding_index":65,"agentic_index":47.5}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"high"}},{"id":"z-ai/glm-5.3","canonical_slug":"z-ai/glm-5.3-20260816","hugging_face_id":"zai-org/GLM-5.3","name":"Z.ai: GLM 5.3","created":1787086655,"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000039","completion":"0.0000048","input_cache_read":"0.000000038"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.3-20260816/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1347,"win_rate":60.7,"rank":5},{"arena":"models","category":"asciiart","elo":1210,"win_rate":46.1,"rank":22},{"arena":"models","category":"codecategories","elo":1313,"win_rate":54.9,"rank":11},{"arena":"models","category":"dataviz","elo":1269,"win_rate":52.7,"rank":23},{"arena":"models","category":"gamedev","elo":1321,"win_rate":52.9,"rank":12},{"arena":"models","category":"svg","elo":1294,"win_rate":56.8,"rank":8},{"arena":"models","category":"uicomponent","elo":1321,"win_rate":57.4,"rank":12},{"arena":"models","category":"website","elo":1304,"win_rate":54.8,"rank":12},{"arena":"agents","category":"htmlslides","elo":1189,"win_rate":39.1,"rank":9},{"arena":"agents","category":"mobileapps","elo":1191,"win_rate":52.2,"rank":17},{"arena":"agents","category":"python-pptxslides","elo":1255,"win_rate":51.2,"rank":8}],"artificial_analysis":{"intelligence_index":44.8,"coding_index":74.8,"agentic_index":53.1}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"z-ai/glm-5.3:batch","canonical_slug":"z-ai/glm-5.3-20260816","hugging_face_id":"zai-org/GLM-5.3","name":"Z.ai: GLM 5.3 (batch)","created":1787086655,"description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000045","completion":"0.000002","input_cache_read":"0.0000001"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.3-20260816/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1347,"win_rate":60.7,"rank":5},{"arena":"models","category":"asciiart","elo":1210,"win_rate":46.1,"rank":22},{"arena":"models","category":"codecategories","elo":1313,"win_rate":54.9,"rank":11},{"arena":"models","category":"dataviz","elo":1269,"win_rate":52.7,"rank":23},{"arena":"models","category":"gamedev","elo":1321,"win_rate":52.9,"rank":12},{"arena":"models","category":"svg","elo":1294,"win_rate":56.8,"rank":8},{"arena":"models","category":"uicomponent","elo":1321,"win_rate":57.4,"rank":12},{"arena":"models","category":"website","elo":1304,"win_rate":54.8,"rank":12},{"arena":"agents","category":"htmlslides","elo":1189,"win_rate":39.1,"rank":9},{"arena":"agents","category":"mobileapps","elo":1191,"win_rate":52.2,"rank":17},{"arena":"agents","category":"python-pptxslides","elo":1255,"win_rate":51.2,"rank":8}],"artificial_analysis":{"intelligence_index":44.8,"coding_index":74.8,"agentic_index":53.1}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"qwen/qwen3.8-27b","canonical_slug":"qwen/qwen3.8-27b-20260814","hugging_face_id":"Qwen/Qwen3.8-27B","name":"Qwen: Qwen3.8 27B","created":1786722910,"description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000000425","completion":"0.00000255","input_cache_read":"0.000000085","input_cache_write":"0.00000053125"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.8-27b-20260814/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":33.7,"coding_index":68.1,"agentic_index":45.8}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["xhigh","medium","low"],"default_effort":"xhigh"}},{"id":"qwen/qwen3.8-2.4t-a95b","canonical_slug":"qwen/qwen3.8-2.4t-a95b-20260812","hugging_face_id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen: Qwen3.8 2.4T A95B","created":1786551702,"description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.8-2.4t-a95b-20260812/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":39.9,"coding_index":71.9,"agentic_index":50.1}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["xhigh","medium","low"],"default_effort":"xhigh"}},{"id":"deepseek/deepseek-v4-pro-0813","canonical_slug":"deepseek/deepseek-v4-pro-20260813","hugging_face_id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek: DeepSeek V4 Pro 0813","created":1786549364,"description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-pro-20260813/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":36,"coding_index":68.8,"agentic_index":41.3}},"reasoning":{"mandatory":false,"supported_efforts":["max","high","low"],"default_effort":"high"}},{"id":"x-ai/grok-4.6","canonical_slug":"x-ai/grok-4.6-20260810","hugging_face_id":null,"name":"SpaceXAI: Grok 4.6","created":1786548957,"description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-4.6-20260810/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1210,"win_rate":47.1,"rank":9},{"arena":"agents","category":"androidnative","elo":1303,"win_rate":61.3,"rank":1},{"arena":"agents","category":"fullstack","elo":1263,"win_rate":54.9,"rank":7},{"arena":"agents","category":"godotgamedev","elo":1276,"win_rate":53,"rank":4},{"arena":"agents","category":"htmlslides","elo":1249,"win_rate":60.1,"rank":3},{"arena":"agents","category":"mobileapps","elo":1233,"win_rate":55.8,"rank":7},{"arena":"agents","category":"python-pptxslides","elo":1242,"win_rate":52.7,"rank":11},{"arena":"agents","category":"webapps","elo":1241,"win_rate":56,"rank":9},{"arena":"models","category":"3d","elo":1282,"win_rate":52.4,"rank":16},{"arena":"models","category":"asciiart","elo":1285,"win_rate":58.3,"rank":7},{"arena":"models","category":"codecategories","elo":1296,"win_rate":52.5,"rank":15},{"arena":"models","category":"dataviz","elo":1286,"win_rate":51.3,"rank":16},{"arena":"models","category":"gamedev","elo":1300,"win_rate":54.1,"rank":18},{"arena":"models","category":"svg","elo":1239,"win_rate":49,"rank":14},{"arena":"models","category":"uicomponent","elo":1297,"win_rate":53.3,"rank":17},{"arena":"models","category":"website","elo":1293,"win_rate":52.3,"rank":16}],"artificial_analysis":{"intelligence_index":44.3,"coding_index":76.8,"agentic_index":53}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["xhigh","high","medium","low"],"default_effort":"high"}},{"id":"nvidia/nemotron-3.5-lightning","canonical_slug":"nvidia/nemotron-3.5-lightning-20260807","hugging_face_id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"NVIDIA: Nemotron 3.5 Lightning","created":1786452751,"description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.0000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3.5-lightning-20260807/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":12.9,"coding_index":26.8,"agentic_index":3.5}},"reasoning":{"mandatory":false}},{"id":"meta/muse-glimmer-30b","canonical_slug":"meta/muse-glimmer-30b-20260810","hugging_face_id":"meta-models/Muse-Glimmer-30B","name":"Meta: Muse Glimmer 30B","created":1786302394,"description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/meta/muse-glimmer-30b-20260810/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":17.5,"coding_index":49,"agentic_index":8.5}},"reasoning":{"mandatory":true,"supported_efforts":["xhigh","high","medium","low"],"default_effort":"medium"}},{"id":"deepseek/deepseek-v4-flash-0731","canonical_slug":"deepseek/deepseek-v4-flash-20260731","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","created":1785478908,"description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000018","completion":"0.00000128","input_cache_read":"0.000000018"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":[],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260731/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1214,"win_rate":48.6,"rank":39},{"arena":"models","category":"asciiart","elo":1097,"win_rate":33.3,"rank":62},{"arena":"models","category":"codecategories","elo":1233,"win_rate":46.4,"rank":46},{"arena":"models","category":"dataviz","elo":1184,"win_rate":41.1,"rank":59},{"arena":"models","category":"gamedev","elo":1214,"win_rate":44.6,"rank":47},{"arena":"models","category":"svg","elo":1186,"win_rate":45.1,"rank":31},{"arena":"models","category":"uicomponent","elo":1239,"win_rate":46.8,"rank":42},{"arena":"models","category":"website","elo":1242,"win_rate":46.7,"rank":45}],"artificial_analysis":{"intelligence_index":34.3,"coding_index":69.1,"agentic_index":41}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"high"}},{"id":"thinkingmachines/inkling-small","canonical_slug":"thinkingmachines/inkling-small-20260730","hugging_face_id":"thinkingmachines/Inkling-Small","name":"Thinking Machines: Inkling Small","created":1785443117,"description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000045","completion":"0.0000012","input_cache_read":"0.0000001"},"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/thinkingmachines/inkling-small-20260730/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":25.7,"coding_index":52.9,"agentic_index":23.5}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","medium","low","minimal","none"],"default_effort":"high"}},{"id":"qwen/qwen3.7-flash","canonical_slug":"qwen/qwen3.7-flash-20260727","hugging_face_id":null,"name":"Qwen: Qwen3.7 Flash","created":1785190561,"description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000013","input_cache_read":"0.000000006","input_cache_write":"0.000000038","overrides":[{"min_prompt_tokens":32000,"prompt":"0.0000001","completion":"0.0000004","input_cache_read":"0.00000002","input_cache_write":"0.000000125"},{"min_prompt_tokens":256000,"prompt":"0.0000002","completion":"0.0000008","input_cache_read":"0.00000004","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.7-flash-20260727/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true}},{"id":"anthropic/claude-opus-5","canonical_slug":"anthropic/claude-opus-5-20260723","hugging_face_id":null,"name":"Anthropic: Claude Opus 5","created":1784912544,"description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-opus-5-20260723/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1261,"win_rate":53.8,"rank":2},{"arena":"agents","category":"androidnative","elo":1269,"win_rate":55.9,"rank":3},{"arena":"agents","category":"fullstack","elo":1305,"win_rate":57.3,"rank":4},{"arena":"agents","category":"godotgamedev","elo":1279,"win_rate":54.2,"rank":3},{"arena":"agents","category":"mobileapps","elo":1348,"win_rate":67.1,"rank":1},{"arena":"agents","category":"python-pptxslides","elo":1267,"win_rate":54.4,"rank":5},{"arena":"agents","category":"webapps","elo":1259,"win_rate":54.5,"rank":7},{"arena":"models","category":"3d","elo":1340,"win_rate":59.8,"rank":7},{"arena":"models","category":"asciiart","elo":1356,"win_rate":67,"rank":1},{"arena":"models","category":"codecategories","elo":1328,"win_rate":57.6,"rank":7},{"arena":"models","category":"dataviz","elo":1334,"win_rate":58.5,"rank":9},{"arena":"models","category":"gamedev","elo":1345,"win_rate":59.4,"rank":8},{"arena":"models","category":"svg","elo":1313,"win_rate":58.6,"rank":3},{"arena":"models","category":"uicomponent","elo":1348,"win_rate":60.4,"rank":5},{"arena":"models","category":"website","elo":1313,"win_rate":56.1,"rank":7}],"artificial_analysis":{"intelligence_index":50.8,"coding_index":78,"agentic_index":56.5}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"inclusionai/ling-3.0-flash","canonical_slug":"inclusionai/ling-3.0-flash-20260723","hugging_face_id":"inclusionAI/Ling-3.0-flash","name":"inclusionAI: Ling 3.0 Flash","created":1784818580,"description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000021","completion":"0.000000063","input_cache_read":"0.0000000042"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/inclusionai/ling-3.0-flash-20260723/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":20.1,"coding_index":50.6,"agentic_index":19.3}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"google/gemini-3.6-flash","canonical_slug":"google/gemini-3.6-flash-20260721","hugging_face_id":null,"name":"Google: Gemini 3.6 Flash","created":1784646733,"description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.6-flash-20260721/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1180,"win_rate":47.8,"rank":13},{"arena":"agents","category":"androidnative","elo":1217,"win_rate":53.7,"rank":11},{"arena":"agents","category":"fullstack","elo":1181,"win_rate":44.3,"rank":21},{"arena":"agents","category":"htmlslides","elo":1148,"win_rate":41.9,"rank":17},{"arena":"agents","category":"mobileapps","elo":1195,"win_rate":50.2,"rank":16},{"arena":"agents","category":"python-pptxslides","elo":1146,"win_rate":38.8,"rank":24},{"arena":"agents","category":"webapps","elo":1187,"win_rate":45.4,"rank":23},{"arena":"models","category":"3d","elo":1279,"win_rate":53.2,"rank":19},{"arena":"models","category":"asciiart","elo":1283,"win_rate":56.8,"rank":8},{"arena":"models","category":"codecategories","elo":1294,"win_rate":53.4,"rank":16},{"arena":"models","category":"dataviz","elo":1294,"win_rate":52.4,"rank":12},{"arena":"models","category":"gamedev","elo":1269,"win_rate":51.1,"rank":27},{"arena":"models","category":"uicomponent","elo":1298,"win_rate":53.7,"rank":15},{"arena":"models","category":"website","elo":1304,"win_rate":55.4,"rank":13}],"artificial_analysis":{"intelligence_index":34,"coding_index":69.2,"agentic_index":29}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["high","medium","low","minimal"],"default_effort":"medium"}},{"id":"google/gemini-3.5-flash-lite","canonical_slug":"google/gemini-3.5-flash-lite-20260721","hugging_face_id":null,"name":"Google: Gemini 3.5 Flash Lite","created":1784646726,"description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","audio":"0.0000003","input_audio_cache":"0.00000003","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.5-flash-lite-20260721/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":22.2,"coding_index":49.3,"agentic_index":14.3}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["high","medium","low","minimal"],"default_effort":"minimal"}},{"id":"thinkingmachines/inkling","canonical_slug":"thinkingmachines/inkling-20260715","hugging_face_id":"thinkingmachines/Inkling","name":"Thinking Machines: Inkling","created":1784325956,"description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.00000405","input_cache_read":"0.00000017"},"top_provider":{"context_length":524288,"max_completion_tokens":471859,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/thinkingmachines/inkling-20260715/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"webapps","elo":1189,"win_rate":48.6,"rank":22},{"arena":"models","category":"3d","elo":1168,"win_rate":40.8,"rank":57},{"arena":"models","category":"asciiart","elo":1097,"win_rate":34,"rank":63},{"arena":"models","category":"codecategories","elo":1209,"win_rate":42.3,"rank":54},{"arena":"models","category":"dataviz","elo":1162,"win_rate":37.3,"rank":73},{"arena":"models","category":"gamedev","elo":1169,"win_rate":37.7,"rank":65},{"arena":"models","category":"svg","elo":1110,"win_rate":34.8,"rank":57},{"arena":"models","category":"uicomponent","elo":1201,"win_rate":40.1,"rank":54},{"arena":"models","category":"website","elo":1228,"win_rate":44.7,"rank":51}],"artificial_analysis":{"intelligence_index":25,"coding_index":52.1,"agentic_index":22.5}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","medium","low","minimal","none"],"default_effort":"high"}},{"id":"moonshotai/kimi-k3","canonical_slug":"moonshotai/kimi-k3-20260715","hugging_face_id":"moonshotai/Kimi-K3","name":"MoonshotAI: Kimi K3","created":1784215858,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.000015","input_cache_read":"0.00000055"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k3-20260715/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1252,"win_rate":55,"rank":3},{"arena":"agents","category":"androidnative","elo":1262,"win_rate":55.4,"rank":5},{"arena":"agents","category":"fullstack","elo":1315,"win_rate":61.8,"rank":2},{"arena":"agents","category":"godotgamedev","elo":1223,"win_rate":48.2,"rank":14},{"arena":"agents","category":"htmlslides","elo":1256,"win_rate":59.6,"rank":2},{"arena":"agents","category":"mobileapps","elo":1263,"win_rate":56.9,"rank":4},{"arena":"agents","category":"python-pptxslides","elo":1273,"win_rate":59.2,"rank":4},{"arena":"agents","category":"webapps","elo":1285,"win_rate":59.1,"rank":4},{"arena":"models","category":"3d","elo":1396,"win_rate":67.6,"rank":4},{"arena":"models","category":"codecategories","elo":1368,"win_rate":62.9,"rank":2},{"arena":"models","category":"dataviz","elo":1354,"win_rate":63.2,"rank":4},{"arena":"models","category":"gamedev","elo":1366,"win_rate":61.1,"rank":4},{"arena":"models","category":"svg","elo":1304,"win_rate":59.7,"rank":4},{"arena":"models","category":"uicomponent","elo":1357,"win_rate":61.5,"rank":4},{"arena":"models","category":"website","elo":1340,"win_rate":59.6,"rank":2}],"artificial_analysis":{"intelligence_index":43.6,"coding_index":76.2,"agentic_index":50}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"moonshotai/kimi-k3:batch","canonical_slug":"moonshotai/kimi-k3-20260715","hugging_face_id":"moonshotai/Kimi-K3","name":"MoonshotAI: Kimi K3 (batch)","created":1784215858,"description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000228","completion":"0.0000114","input_cache_read":"0.000000228"},"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k3-20260715/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1252,"win_rate":55,"rank":3},{"arena":"agents","category":"androidnative","elo":1262,"win_rate":55.4,"rank":5},{"arena":"agents","category":"fullstack","elo":1315,"win_rate":61.8,"rank":2},{"arena":"agents","category":"godotgamedev","elo":1223,"win_rate":48.2,"rank":14},{"arena":"agents","category":"htmlslides","elo":1256,"win_rate":59.6,"rank":2},{"arena":"agents","category":"mobileapps","elo":1263,"win_rate":56.9,"rank":4},{"arena":"agents","category":"python-pptxslides","elo":1273,"win_rate":59.2,"rank":4},{"arena":"agents","category":"webapps","elo":1285,"win_rate":59.1,"rank":4},{"arena":"models","category":"3d","elo":1396,"win_rate":67.6,"rank":4},{"arena":"models","category":"codecategories","elo":1368,"win_rate":62.9,"rank":2},{"arena":"models","category":"dataviz","elo":1354,"win_rate":63.2,"rank":4},{"arena":"models","category":"gamedev","elo":1366,"win_rate":61.1,"rank":4},{"arena":"models","category":"svg","elo":1304,"win_rate":59.7,"rank":4},{"arena":"models","category":"uicomponent","elo":1357,"win_rate":61.5,"rank":4},{"arena":"models","category":"website","elo":1340,"win_rate":59.6,"rank":2}],"artificial_analysis":{"intelligence_index":43.6,"coding_index":76.2,"agentic_index":50}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","high","low"],"default_effort":"max"}},{"id":"openai/gpt-5.6-luna","canonical_slug":"openai/gpt-5.6-luna-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Luna","created":1783590864,"description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000012","web_search":"0.01","input_cache_read":"0.00000002","input_cache_write":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000004","completion":"0.0000018","input_cache_read":"0.00000004","input_cache_write":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-luna-20260709/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":37.3,"coding_index":71.4,"agentic_index":42.1}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"openai/gpt-5.6-terra","canonical_slug":"openai/gpt-5.6-terra-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Terra","created":1783590857,"description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000018","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-terra-20260709/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":42.1,"coding_index":76.7,"agentic_index":43.2}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"openai/gpt-5.6-sol","canonical_slug":"openai/gpt-5.6-sol-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Sol","created":1783590850,"description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.6-sol-20260709/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":47,"coding_index":77.4,"agentic_index":50.2}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"anthropic/claude-sonnet-5","canonical_slug":"anthropic/claude-sonnet-5-20260630","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5","created":1782843083,"description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-sonnet-5-20260630/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1226,"win_rate":52.3,"rank":7},{"arena":"agents","category":"androidnative","elo":1237,"win_rate":54.4,"rank":9},{"arena":"agents","category":"fullstack","elo":1248,"win_rate":53.7,"rank":10},{"arena":"agents","category":"godotgamedev","elo":1225,"win_rate":54.4,"rank":13},{"arena":"agents","category":"htmlslides","elo":1217,"win_rate":53.7,"rank":5},{"arena":"agents","category":"mobileapps","elo":1211,"win_rate":50.9,"rank":13},{"arena":"agents","category":"python-pptxslides","elo":1216,"win_rate":50.1,"rank":13},{"arena":"agents","category":"webapps","elo":1237,"win_rate":53.3,"rank":10},{"arena":"models","category":"3d","elo":1267,"win_rate":54.2,"rank":22},{"arena":"models","category":"asciiart","elo":1214,"win_rate":51.6,"rank":20},{"arena":"models","category":"codecategories","elo":1281,"win_rate":53,"rank":20},{"arena":"models","category":"dataviz","elo":1252,"win_rate":52.7,"rank":31},{"arena":"models","category":"gamedev","elo":1287,"win_rate":52.5,"rank":20},{"arena":"models","category":"svg","elo":1192,"win_rate":49.8,"rank":29},{"arena":"models","category":"uicomponent","elo":1283,"win_rate":54,"rank":24},{"arena":"models","category":"website","elo":1279,"win_rate":52.7,"rank":22}],"artificial_analysis":{"intelligence_index":38.2,"coding_index":71.5,"agentic_index":43.6}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"z-ai/glm-5.2","canonical_slug":"z-ai/glm-5.2-20260616","hugging_face_id":"zai-org/GLM-5.2","name":"Z.ai: GLM 5.2","created":1781631930,"description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.000007","input_cache_read":"0.000000059"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-5.2-20260616/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1205,"win_rate":49,"rank":10},{"arena":"agents","category":"androidnative","elo":1193,"win_rate":53.2,"rank":16},{"arena":"agents","category":"fullstack","elo":1241,"win_rate":60.5,"rank":12},{"arena":"agents","category":"godotgamedev","elo":1142,"win_rate":40.1,"rank":22},{"arena":"agents","category":"htmlslides","elo":1175,"win_rate":49.5,"rank":13},{"arena":"agents","category":"mobileapps","elo":1156,"win_rate":48.1,"rank":25},{"arena":"agents","category":"python-pptxslides","elo":1194,"win_rate":48.7,"rank":16},{"arena":"agents","category":"webapps","elo":1215,"win_rate":56,"rank":13},{"arena":"models","category":"3d","elo":1294,"win_rate":53.4,"rank":15},{"arena":"models","category":"asciiart","elo":1200,"win_rate":46.2,"rank":26},{"arena":"models","category":"codecategories","elo":1293,"win_rate":50.5,"rank":18},{"arena":"models","category":"dataviz","elo":1290,"win_rate":52.1,"rank":13},{"arena":"models","category":"gamedev","elo":1275,"win_rate":49.6,"rank":23},{"arena":"models","category":"svg","elo":1217,"win_rate":49.1,"rank":21},{"arena":"models","category":"uicomponent","elo":1298,"win_rate":55.2,"rank":16},{"arena":"models","category":"website","elo":1290,"win_rate":49.6,"rank":18}],"artificial_analysis":{"intelligence_index":33.7,"coding_index":68.8,"agentic_index":38.4}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["xhigh","high"],"default_effort":"high"}},{"id":"nvidia/nemotron-3.5-content-safety","canonical_slug":"nvidia/nemotron-3.5-content-safety-20260604","hugging_face_id":"nvidia/Nemotron-3.5-Content-Safety","name":"NVIDIA: Nemotron 3.5 Content Safety","created":1780581864,"description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3.5-content-safety-20260604/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"nvidia/nemotron-3-ultra-550b-a55b","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","name":"NVIDIA: Nemotron 3 Ultra","created":1780551208,"description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-ultra-550b-a55b-20260604/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1138,"win_rate":39.2,"rank":69},{"arena":"models","category":"asciiart","elo":1104,"win_rate":38.2,"rank":59},{"arena":"models","category":"codecategories","elo":1149,"win_rate":36.1,"rank":81},{"arena":"models","category":"dataviz","elo":1143,"win_rate":37.3,"rank":79},{"arena":"models","category":"gamedev","elo":1144,"win_rate":36.5,"rank":74},{"arena":"models","category":"svg","elo":1068,"win_rate":32.5,"rank":69},{"arena":"models","category":"uicomponent","elo":1145,"win_rate":36.4,"rank":75},{"arena":"models","category":"website","elo":1145,"win_rate":34.9,"rank":90}],"artificial_analysis":{"intelligence_index":22.9,"coding_index":49.3,"agentic_index":20.1}},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true,"supported_efforts":["high","medium"],"default_effort":"high"}},{"id":"qwen/qwen3.7-plus","canonical_slug":"qwen/qwen3.7-plus-20260602","hugging_face_id":null,"name":"Qwen: Qwen3.7 Plus","created":1780491783,"description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.00000128","input_cache_read":"0.000000064","input_cache_write":"0.0000004","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000096","completion":"0.00000384","input_cache_read":"0.000000192","input_cache_write":"0.0000012"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.7-plus-20260602/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1243,"win_rate":48.4,"rank":32},{"arena":"models","category":"asciiart","elo":1147,"win_rate":41.1,"rank":46},{"arena":"models","category":"codecategories","elo":1273,"win_rate":50.5,"rank":26},{"arena":"models","category":"dataviz","elo":1280,"win_rate":51.1,"rank":20},{"arena":"models","category":"gamedev","elo":1266,"win_rate":49.6,"rank":32},{"arena":"models","category":"svg","elo":1207,"win_rate":46.8,"rank":23},{"arena":"models","category":"uicomponent","elo":1264,"win_rate":49.1,"rank":32},{"arena":"models","category":"website","elo":1275,"win_rate":51,"rank":25}],"artificial_analysis":{"intelligence_index":25.2,"coding_index":55.9,"agentic_index":17.5}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"minimax/minimax-m3","canonical_slug":"minimax/minimax-m3-20260531","hugging_face_id":"MiniMaxAI/Minimax-M3","name":"MiniMax: MiniMax M3","created":1780245374,"description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006"},"top_provider":{"context_length":524288,"max_completion_tokens":512000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/minimax-m3-20260531/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1158,"win_rate":44.6,"rank":19},{"arena":"agents","category":"androidnative","elo":1171,"win_rate":44.3,"rank":23},{"arena":"agents","category":"fullstack","elo":1196,"win_rate":47.3,"rank":19},{"arena":"agents","category":"htmlslides","elo":1180,"win_rate":46.6,"rank":12},{"arena":"agents","category":"mobileapps","elo":1167,"win_rate":46,"rank":23},{"arena":"agents","category":"python-pptxslides","elo":1204,"win_rate":46.8,"rank":14},{"arena":"agents","category":"webapps","elo":1192,"win_rate":47.1,"rank":21},{"arena":"models","category":"3d","elo":1214,"win_rate":50.2,"rank":41},{"arena":"models","category":"asciiart","elo":1169,"win_rate":46.8,"rank":35},{"arena":"models","category":"codecategories","elo":1252,"win_rate":50.9,"rank":37},{"arena":"models","category":"dataviz","elo":1239,"win_rate":51.3,"rank":37},{"arena":"models","category":"gamedev","elo":1218,"win_rate":46,"rank":44},{"arena":"models","category":"svg","elo":1162,"win_rate":45.2,"rank":39},{"arena":"models","category":"uicomponent","elo":1250,"win_rate":50.9,"rank":37},{"arena":"models","category":"website","elo":1262,"win_rate":51.9,"rank":34}],"artificial_analysis":{"intelligence_index":29.2,"coding_index":58.6,"agentic_index":29.5}},"reasoning":{"mandatory":false}},{"id":"anthropic/claude-opus-4.8","canonical_slug":"anthropic/claude-4.8-opus-20260528","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.8","created":1779905091,"description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.8-opus-20260528/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1238,"win_rate":55.8,"rank":4},{"arena":"agents","category":"agentichtmlslides","elo":1227,"win_rate":55.6,"rank":3},{"arena":"agents","category":"agenticslides","elo":1294,"win_rate":64.8,"rank":1},{"arena":"agents","category":"agenticslides(html)","elo":1230,"win_rate":56,"rank":3},{"arena":"agents","category":"agenticslides(python-pptx)","elo":1310,"win_rate":68.9,"rank":1},{"arena":"agents","category":"androidnative","elo":1247,"win_rate":56.4,"rank":8},{"arena":"agents","category":"fullstack","elo":1248,"win_rate":55.3,"rank":9},{"arena":"agents","category":"godotgamedev","elo":1231,"win_rate":55.8,"rank":10},{"arena":"agents","category":"htmlslides","elo":1220,"win_rate":58,"rank":4},{"arena":"agents","category":"mobileapps","elo":1220,"win_rate":55.4,"rank":9},{"arena":"agents","category":"pptxslides","elo":1306,"win_rate":67.9,"rank":1},{"arena":"agents","category":"python-pptxslides","elo":1298,"win_rate":65.9,"rank":2},{"arena":"agents","category":"webapps","elo":1224,"win_rate":51.4,"rank":12},{"arena":"models","category":"3d","elo":1227,"win_rate":51.8,"rank":36},{"arena":"models","category":"asciiart","elo":1281,"win_rate":61.8,"rank":9},{"arena":"models","category":"codecategories","elo":1258,"win_rate":52.6,"rank":35},{"arena":"models","category":"dataviz","elo":1243,"win_rate":53,"rank":33},{"arena":"models","category":"gamedev","elo":1269,"win_rate":53.2,"rank":25},{"arena":"models","category":"svg","elo":1192,"win_rate":51.6,"rank":28},{"arena":"models","category":"uicomponent","elo":1261,"win_rate":53.3,"rank":34},{"arena":"models","category":"website","elo":1261,"win_rate":53.1,"rank":35}],"artificial_analysis":{"intelligence_index":41.8,"coding_index":74.3,"agentic_index":41.9}},"reasoning":{"mandatory":false,"default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"qwen/qwen3.7-max","canonical_slug":"qwen/qwen3.7-max-20260520","hugging_face_id":null,"name":"Qwen: Qwen3.7 Max","created":1779376861,"description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000001475","completion":"0.000004425","input_cache_read":"0.000000295","input_cache_write":"0.00000184375"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.7-max-20260520/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1161,"win_rate":46.6,"rank":18},{"arena":"agents","category":"androidnative","elo":1174,"win_rate":48.9,"rank":22},{"arena":"agents","category":"fullstack","elo":1190,"win_rate":47.7,"rank":20},{"arena":"agents","category":"godotgamedev","elo":1187,"win_rate":55.7,"rank":16},{"arena":"agents","category":"htmlslides","elo":1158,"win_rate":43.6,"rank":15},{"arena":"agents","category":"mobileapps","elo":1147,"win_rate":44.8,"rank":27},{"arena":"agents","category":"python-pptxslides","elo":1194,"win_rate":47.2,"rank":17},{"arena":"agents","category":"webapps","elo":1202,"win_rate":49.4,"rank":17},{"arena":"models","category":"3d","elo":1272,"win_rate":54.7,"rank":21},{"arena":"models","category":"asciiart","elo":1219,"win_rate":53.3,"rank":19},{"arena":"models","category":"codecategories","elo":1279,"win_rate":54.3,"rank":24},{"arena":"models","category":"dataviz","elo":1280,"win_rate":53,"rank":19},{"arena":"models","category":"gamedev","elo":1273,"win_rate":53.6,"rank":24},{"arena":"models","category":"svg","elo":1216,"win_rate":55.6,"rank":22},{"arena":"models","category":"uicomponent","elo":1274,"win_rate":53.6,"rank":28},{"arena":"models","category":"website","elo":1277,"win_rate":54.6,"rank":23}],"artificial_analysis":{"intelligence_index":29.5,"coding_index":66,"agentic_index":22.5}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"google/gemini-3.5-flash","canonical_slug":"google/gemini-3.5-flash-20260519","hugging_face_id":null,"name":"Google: Gemini 3.5 Flash","created":1779193800,"description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000015","completion":"0.000009","image":"0.0000015","audio":"0.000003","input_audio_cache":"0.0000003","web_search":"0.014","internal_reasoning":"0.000009","input_cache_read":"0.00000015","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-01","expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.5-flash-20260519/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1180,"win_rate":55.4,"rank":12},{"arena":"agents","category":"agentichtmlslides","elo":1162,"win_rate":45.8,"rank":6},{"arena":"agents","category":"agenticslides","elo":1244,"win_rate":57.5,"rank":3},{"arena":"agents","category":"agenticslides(html)","elo":1162,"win_rate":45.7,"rank":6},{"arena":"agents","category":"agenticslides(python-pptx)","elo":1242,"win_rate":57.8,"rank":2},{"arena":"agents","category":"androidnative","elo":1199,"win_rate":57.7,"rank":14},{"arena":"agents","category":"fullstack","elo":1197,"win_rate":56.2,"rank":18},{"arena":"agents","category":"godotgamedev","elo":1098,"win_rate":43,"rank":30},{"arena":"agents","category":"htmlslides","elo":1142,"win_rate":46.5,"rank":19},{"arena":"agents","category":"mobileapps","elo":1173,"win_rate":51.2,"rank":21},{"arena":"agents","category":"pptxslides","elo":1244,"win_rate":57.7,"rank":2},{"arena":"agents","category":"python-pptxslides","elo":1247,"win_rate":57.4,"rank":9},{"arena":"agents","category":"webapps","elo":1196,"win_rate":52.6,"rank":20},{"arena":"models","category":"3d","elo":1250,"win_rate":56.3,"rank":29},{"arena":"models","category":"asciiart","elo":1259,"win_rate":58.9,"rank":17},{"arena":"models","category":"codecategories","elo":1269,"win_rate":54.3,"rank":30},{"arena":"models","category":"dataviz","elo":1239,"win_rate":52.9,"rank":35},{"arena":"models","category":"gamedev","elo":1269,"win_rate":53,"rank":26},{"arena":"models","category":"svg","elo":1253,"win_rate":58.5,"rank":13},{"arena":"models","category":"uicomponent","elo":1277,"win_rate":54.8,"rank":26},{"arena":"models","category":"website","elo":1270,"win_rate":53.8,"rank":29}],"artificial_analysis":{"intelligence_index":33.6,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"default_enabled":true,"supported_efforts":["high","medium","low","minimal"],"default_effort":"medium"}},{"id":"google/gemini-3.1-flash-lite","canonical_slug":"google/gemini-3.1-flash-lite-20260507","hugging_face_id":null,"name":"Google: Gemini 3.1 Flash Lite","created":1778168828,"description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemini-3.1-flash-lite-20260507/endpoints"},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["high","medium","low","minimal"],"default_effort":"minimal"}},{"id":"qwen/qwen3.6-35b-a3b","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","hugging_face_id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen: Qwen3.6 35B A3B","created":1777260255,"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.00000095","input_cache_read":"0.0000001"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-35b-a3b-20260415/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":18.2,"coding_index":41.9,"agentic_index":13.1}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"qwen/qwen3.6-27b","canonical_slug":"qwen/qwen3.6-27b-20260422","hugging_face_id":"Qwen/Qwen3.6-27B","name":"Qwen: Qwen3.6 27B","created":1777255064,"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.0000032"},"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-27b-20260422/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":21.4,"coding_index":53.7,"agentic_index":18.5}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"openai/gpt-5.5","canonical_slug":"openai/gpt-5.5-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5","created":1777051893,"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.5-20260423/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1179,"win_rate":52.4,"rank":14},{"arena":"agents","category":"agentichtmlslides","elo":1084,"win_rate":34.2,"rank":8},{"arena":"agents","category":"agenticslides","elo":1150,"win_rate":43.5,"rank":6},{"arena":"agents","category":"agenticslides(html)","elo":1077,"win_rate":33.2,"rank":8},{"arena":"agents","category":"agenticslides(python-pptx)","elo":1155,"win_rate":45.2,"rank":6},{"arena":"agents","category":"androidnative","elo":1177,"win_rate":50.9,"rank":19},{"arena":"agents","category":"fullstack","elo":1088,"win_rate":43,"rank":29},{"arena":"agents","category":"godotgamedev","elo":1173,"win_rate":52.3,"rank":18},{"arena":"agents","category":"htmlslides","elo":1067,"win_rate":35.6,"rank":22},{"arena":"agents","category":"mobileapps","elo":1139,"win_rate":50.4,"rank":30},{"arena":"agents","category":"pptxslides","elo":1157,"win_rate":45.3,"rank":6},{"arena":"agents","category":"python-pptxslides","elo":1152,"win_rate":43.3,"rank":23},{"arena":"agents","category":"webapps","elo":1110,"win_rate":42.6,"rank":33},{"arena":"models","category":"3d","elo":1203,"win_rate":49.6,"rank":44},{"arena":"models","category":"asciiart","elo":1261,"win_rate":58.4,"rank":14},{"arena":"models","category":"codecategories","elo":1264,"win_rate":53.4,"rank":32},{"arena":"models","category":"dataviz","elo":1269,"win_rate":55.9,"rank":24},{"arena":"models","category":"gamedev","elo":1300,"win_rate":57.8,"rank":17},{"arena":"models","category":"svg","elo":1234,"win_rate":55.3,"rank":16},{"arena":"models","category":"uicomponent","elo":1263,"win_rate":53.4,"rank":33},{"arena":"models","category":"website","elo":1261,"win_rate":52.7,"rank":36}],"artificial_analysis":{"intelligence_index":38.4,"coding_index":74.9,"agentic_index":36.4}},"reasoning":{"mandatory":false,"default_enabled":true,"supported_efforts":["xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"deepseek/deepseek-v4-pro","canonical_slug":"deepseek/deepseek-v4-pro-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek: DeepSeek V4 Pro 0423","created":1777000679,"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174"},"top_provider":{"context_length":1024000,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-pro-20260423/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"fullstack","elo":948,"win_rate":22.1,"rank":43},{"arena":"agents","category":"godotgamedev","elo":1059,"win_rate":34,"rank":32},{"arena":"agents","category":"webapps","elo":1000,"win_rate":26.4,"rank":42},{"arena":"models","category":"3d","elo":1259,"win_rate":56.8,"rank":25},{"arena":"models","category":"asciiart","elo":1155,"win_rate":46.4,"rank":41},{"arena":"models","category":"codecategories","elo":1244,"win_rate":51.9,"rank":41},{"arena":"models","category":"dataviz","elo":1206,"win_rate":48.7,"rank":52},{"arena":"models","category":"gamedev","elo":1236,"win_rate":52.3,"rank":39},{"arena":"models","category":"svg","elo":1139,"win_rate":45.4,"rank":49},{"arena":"models","category":"uicomponent","elo":1225,"win_rate":50.6,"rank":46},{"arena":"models","category":"website","elo":1239,"win_rate":50.5,"rank":46}],"artificial_analysis":{"intelligence_index":30.4,"coding_index":59.4,"agentic_index":26.3}},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"],"default_effort":"high"}},{"id":"deepseek/deepseek-v4-flash","canonical_slug":"deepseek/deepseek-v4-flash-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek: DeepSeek V4 Flash 0423","created":1777000666,"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000128","input_cache_read":"0.00000003"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260423/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1194,"win_rate":49.3,"rank":48},{"arena":"models","category":"asciiart","elo":1117,"win_rate":42.8,"rank":57},{"arena":"models","category":"codecategories","elo":1209,"win_rate":48.9,"rank":53},{"arena":"models","category":"dataviz","elo":1130,"win_rate":40.7,"rank":85},{"arena":"models","category":"gamedev","elo":1204,"win_rate":50.2,"rank":50},{"arena":"models","category":"svg","elo":1155,"win_rate":48.4,"rank":41},{"arena":"models","category":"uicomponent","elo":1168,"win_rate":44.7,"rank":67},{"arena":"models","category":"website","elo":1211,"win_rate":49.1,"rank":55}],"artificial_analysis":{"intelligence_index":24.4,"coding_index":52,"agentic_index":26.3}},"reasoning":{"mandatory":false,"supported_efforts":["xhigh","high"],"default_effort":"high"}},{"id":"moonshotai/kimi-k2.6","canonical_slug":"moonshotai/kimi-k2.6-20260420","hugging_face_id":"moonshotai/Kimi-K2.6","name":"MoonshotAI: Kimi K2.6","created":1776699402,"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000004344","completion":"0.00000245","input_cache_read":"0.0000001077"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2.6-20260420/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1141,"win_rate":47.5,"rank":20},{"arena":"agents","category":"agentichtmlslides","elo":1248,"win_rate":59,"rank":2},{"arena":"agents","category":"agenticslides","elo":1187,"win_rate":45.8,"rank":4},{"arena":"agents","category":"agenticslides(html)","elo":1252,"win_rate":59.2,"rank":2},{"arena":"agents","category":"agenticslides(python-pptx)","elo":1186,"win_rate":45.5,"rank":4},{"arena":"agents","category":"androidnative","elo":1174,"win_rate":49.1,"rank":21},{"arena":"agents","category":"fullstack","elo":1156,"win_rate":49.4,"rank":24},{"arena":"agents","category":"godotgamedev","elo":1120,"win_rate":46.8,"rank":28},{"arena":"agents","category":"htmlslides","elo":1184,"win_rate":52.3,"rank":10},{"arena":"agents","category":"mobileapps","elo":1153,"win_rate":47.8,"rank":26},{"arena":"agents","category":"pptxslides","elo":1181,"win_rate":44.3,"rank":4},{"arena":"agents","category":"python-pptxslides","elo":1180,"win_rate":42.1,"rank":18},{"arena":"agents","category":"webapps","elo":1268,"win_rate":59.3,"rank":6},{"arena":"models","category":"3d","elo":1273,"win_rate":57.6,"rank":20},{"arena":"models","category":"asciiart","elo":1170,"win_rate":47.5,"rank":34},{"arena":"models","category":"codecategories","elo":1275,"win_rate":54.8,"rank":25},{"arena":"models","category":"dataviz","elo":1253,"win_rate":52,"rank":30},{"arena":"models","category":"gamedev","elo":1256,"win_rate":54.9,"rank":33},{"arena":"models","category":"svg","elo":1181,"win_rate":51.2,"rank":32},{"arena":"models","category":"uicomponent","elo":1269,"win_rate":55.6,"rank":29},{"arena":"models","category":"website","elo":1274,"win_rate":54.2,"rank":27}],"artificial_analysis":{"intelligence_index":27,"coding_index":61.8,"agentic_index":20.5}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"anthropic/claude-opus-4.7","canonical_slug":"anthropic/claude-4.7-opus-20260416","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.7","created":1776351100,"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.7-opus-20260416/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":40.7,"coding_index":73.6,"agentic_index":38.6}},"reasoning":{"mandatory":false,"default_enabled":false,"supported_efforts":["max","xhigh","high","medium","low"],"default_effort":"high"}},{"id":"google/gemma-4-26b-a4b-it","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","hugging_face_id":"google/gemma-4-26B-A4B-it","name":"Google: Gemma 4 26B A4B ","created":1775227989,"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.0000000675","completion":"0.000000225","input_cache_read":"0.0000000375"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":16.7,"coding_index":39.3,"agentic_index":2.9}},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"google/gemma-4-31b-it","canonical_slug":"google/gemma-4-31b-it-20260402","hugging_face_id":"google/gemma-4-31B-it","name":"Google: Gemma 4 31B","created":1775148486,"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-4-31b-it-20260402/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":14.7,"coding_index":43.4,"agentic_index":4.2}},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"openai/gpt-5.4-nano","canonical_slug":"openai/gpt-5.4-nano-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Nano","created":1773748187,"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.00000125","web_search":"0.01","input_cache_read":"0.00000002"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-nano-20260317/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":20.7,"coding_index":56.1,"agentic_index":16}},"reasoning":{"mandatory":false,"default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"openai/gpt-5.4-mini","canonical_slug":"openai/gpt-5.4-mini-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Mini","created":1773748178,"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000045","web_search":"0.01","input_cache_read":"0.000000075"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-mini-20260317/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":24.1,"coding_index":56.1,"agentic_index":17.9}},"reasoning":{"mandatory":false,"default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"mistralai/mistral-small-2603","canonical_slug":"mistralai/mistral-small-2603","hugging_face_id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral: Mistral Small 4","created":1773695685,"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-2603/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":11.3,"coding_index":26.6,"agentic_index":0.8}},"reasoning":{"mandatory":false,"default_enabled":false,"supported_efforts":["high","none"],"default_effort":"high"}},{"id":"nvidia/nemotron-3-super-120b-a12b","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"NVIDIA: Nemotron 3 Super","created":1773245239,"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.00000045"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":12.8,"coding_index":37.7,"agentic_index":1.7}},"reasoning":{"mandatory":false,"default_enabled":true,"supports_max_tokens":true,"supported_efforts":["medium","low"],"default_effort":"medium"}},{"id":"qwen/qwen3.5-9b","canonical_slug":"qwen/qwen3.5-9b-20260310","hugging_face_id":"Qwen/Qwen3.5-9B","name":"Qwen: Qwen3.5-9B","created":1773152396,"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.00000015"},"top_provider":{"context_length":256000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-21","links":{"details":"/api/v1/models/qwen/qwen3.5-9b-20260310/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":13.3,"coding_index":23.5,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"openai/gpt-5.4","canonical_slug":"openai/gpt-5.4-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4","created":1772734352,"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000005","completion":"0.0000225","input_cache_read":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-20260305/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1107,"win_rate":42.3,"rank":81},{"arena":"models","category":"asciiart","elo":1204,"win_rate":55.4,"rank":25},{"arena":"models","category":"codecategories","elo":1215,"win_rate":52.5,"rank":51},{"arena":"models","category":"dataviz","elo":1238,"win_rate":56.6,"rank":38},{"arena":"models","category":"gamedev","elo":1246,"win_rate":57.6,"rank":36},{"arena":"models","category":"svg","elo":1194,"win_rate":57.8,"rank":27},{"arena":"models","category":"uicomponent","elo":1243,"win_rate":57.4,"rank":38},{"arena":"models","category":"website","elo":1224,"win_rate":52.5,"rank":54},{"arena":"agents","category":"androidnative","elo":1046,"win_rate":47.4,"rank":36},{"arena":"agents","category":"fullstack","elo":1016,"win_rate":40.9,"rank":39},{"arena":"agents","category":"godotgamedev","elo":1135,"win_rate":46.9,"rank":25},{"arena":"agents","category":"mobileapps","elo":1084,"win_rate":46.1,"rank":42},{"arena":"agents","category":"webapps","elo":1052,"win_rate":40.3,"rank":38}],"artificial_analysis":{"intelligence_index":39,"coding_index":71.1,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":false,"supported_efforts":["xhigh","high","medium","low","none"],"default_effort":"medium"}},{"id":"qwen/qwen3.5-35b-a3b","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","hugging_face_id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen: Qwen3.5-35B-A3B","created":1772053822,"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.00000075","input_cache_read":"0.00000004"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-35b-a3b-20260224/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":19.3,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"qwen/qwen3.5-27b","canonical_slug":"qwen/qwen3.5-27b-20260224","hugging_face_id":"Qwen/Qwen3.5-27B","name":"Qwen: Qwen3.5-27B","created":1772053810,"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.0000026"},"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-27b-20260224/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":22.9,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"anthropic/claude-sonnet-4.6","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.6","created":1771342990,"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-sonnet-20260217/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"agenticgamedev","elo":1177,"win_rate":52.2,"rank":16},{"arena":"agents","category":"androidnative","elo":1211,"win_rate":62,"rank":13},{"arena":"agents","category":"fullstack","elo":1213,"win_rate":64.3,"rank":14},{"arena":"agents","category":"godotgamedev","elo":1226,"win_rate":60.6,"rank":12},{"arena":"agents","category":"mobileapps","elo":1209,"win_rate":63.6,"rank":14},{"arena":"agents","category":"webapps","elo":1183,"win_rate":57,"rank":24},{"arena":"models","category":"3d","elo":1240,"win_rate":57.6,"rank":33},{"arena":"models","category":"asciiart","elo":1242,"win_rate":60.1,"rank":18},{"arena":"models","category":"codecategories","elo":1280,"win_rate":59.4,"rank":22},{"arena":"models","category":"dataviz","elo":1281,"win_rate":58.2,"rank":18},{"arena":"models","category":"gamedev","elo":1266,"win_rate":58.8,"rank":30},{"arena":"models","category":"svg","elo":1196,"win_rate":58.7,"rank":25},{"arena":"models","category":"uicomponent","elo":1275,"win_rate":58.3,"rank":27},{"arena":"models","category":"website","elo":1289,"win_rate":60,"rank":19}],"artificial_analysis":{"intelligence_index":30.1,"coding_index":63,"agentic_index":31.8}},"reasoning":{"mandatory":false,"supported_efforts":["max","high","medium","low"],"default_effort":"medium"}},{"id":"qwen/qwen3.5-397b-a17b","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","hugging_face_id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen: Qwen3.5 397B A17B","created":1771223018,"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000035","input_cache_read":"0.000000225"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-397b-a17b-20260216/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1166,"win_rate":56.7,"rank":58},{"arena":"models","category":"codecategories","elo":1184,"win_rate":52.6,"rank":62},{"arena":"models","category":"dataviz","elo":1178,"win_rate":53.2,"rank":63},{"arena":"models","category":"gamedev","elo":1146,"win_rate":50.2,"rank":72},{"arena":"models","category":"svg","elo":1139,"win_rate":56.1,"rank":50},{"arena":"models","category":"uicomponent","elo":1167,"win_rate":51.4,"rank":69},{"arena":"models","category":"website","elo":1195,"win_rate":52.5,"rank":63}],"artificial_analysis":{"intelligence_index":21.4,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"moonshotai/kimi-k2.5","canonical_slug":"moonshotai/kimi-k2.5-0127","hugging_face_id":"moonshotai/Kimi-K2.5","name":"MoonshotAI: Kimi K2.5","created":1769487076,"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000025","input_cache_read":"0.00000015"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2.5-0127/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"androidnative","elo":1133,"win_rate":57.8,"rank":26},{"arena":"agents","category":"fullstack","elo":1114,"win_rate":54.5,"rank":28},{"arena":"agents","category":"godotgamedev","elo":1219,"win_rate":59.8,"rank":15},{"arena":"agents","category":"mobileapps","elo":1144,"win_rate":54.7,"rank":28},{"arena":"agents","category":"webapps","elo":1129,"win_rate":51.6,"rank":29},{"arena":"models","category":"3d","elo":1210,"win_rate":53.1,"rank":42},{"arena":"models","category":"asciiart","elo":1171,"win_rate":46.4,"rank":32},{"arena":"models","category":"codecategories","elo":1240,"win_rate":54.1,"rank":42},{"arena":"models","category":"dataviz","elo":1222,"win_rate":51.3,"rank":47},{"arena":"models","category":"gamedev","elo":1212,"win_rate":53.4,"rank":48},{"arena":"models","category":"svg","elo":1149,"win_rate":48.4,"rank":44},{"arena":"models","category":"uicomponent","elo":1240,"win_rate":53.6,"rank":41},{"arena":"models","category":"website","elo":1252,"win_rate":55.2,"rank":41}],"artificial_analysis":{"intelligence_index":23.5,"coding_index":46.8,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"z-ai/glm-4.7","canonical_slug":"z-ai/glm-4.7-20251222","hugging_face_id":"zai-org/GLM-4.7","name":"Z.ai: GLM 4.7","created":1766378014,"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-12-31","links":{"details":"/api/v1/models/z-ai/glm-4.7-20251222/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"androidnative","elo":1128,"win_rate":56.2,"rank":27},{"arena":"agents","category":"fullstack","elo":1050,"win_rate":45.1,"rank":35},{"arena":"agents","category":"godotgamedev","elo":1058,"win_rate":35.8,"rank":33},{"arena":"agents","category":"mobileapps","elo":1102,"win_rate":48.9,"rank":35},{"arena":"models","category":"3d","elo":1196,"win_rate":54.3,"rank":47},{"arena":"models","category":"asciiart","elo":1170,"win_rate":48.2,"rank":33},{"arena":"models","category":"codecategories","elo":1221,"win_rate":54.8,"rank":50},{"arena":"models","category":"dataviz","elo":1199,"win_rate":51.2,"rank":56},{"arena":"models","category":"gamedev","elo":1194,"win_rate":55.2,"rank":55},{"arena":"models","category":"svg","elo":1142,"win_rate":54.3,"rank":48},{"arena":"models","category":"uicomponent","elo":1202,"win_rate":51.1,"rank":53},{"arena":"models","category":"website","elo":1230,"win_rate":55.3,"rank":50}],"artificial_analysis":{"intelligence_index":22.2,"coding_index":45.3,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"deepseek/deepseek-v3.2","canonical_slug":"deepseek/deepseek-v3.2-20251201","hugging_face_id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek: DeepSeek V3.2","created":1764594642,"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000259","completion":"0.0000008","input_cache_read":"0.00000014"},"top_provider":{"context_length":163840,"max_completion_tokens":147456,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-20251201/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1138,"win_rate":49.4,"rank":68},{"arena":"models","category":"asciiart","elo":1089,"win_rate":40.5,"rank":64},{"arena":"models","category":"codecategories","elo":1168,"win_rate":49.3,"rank":74},{"arena":"models","category":"dataviz","elo":1162,"win_rate":48.2,"rank":71},{"arena":"models","category":"gamedev","elo":1137,"win_rate":46.5,"rank":80},{"arena":"models","category":"svg","elo":1031,"win_rate":40.8,"rank":77},{"arena":"models","category":"uicomponent","elo":1152,"win_rate":46.8,"rank":72},{"arena":"models","category":"website","elo":1179,"win_rate":50.2,"rank":74}],"artificial_analysis":{"intelligence_index":21.5,"coding_index":44.2,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":false}},{"id":"anthropic/claude-haiku-4.5","canonical_slug":"anthropic/claude-4.5-haiku-20251001","hugging_face_id":"","name":"Anthropic: Claude Haiku 4.5","created":1760547638,"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-haiku-20251001/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1077,"win_rate":41.1,"rank":93},{"arena":"models","category":"asciiart","elo":1145,"win_rate":49.3,"rank":48},{"arena":"models","category":"codecategories","elo":1119,"win_rate":44.9,"rank":91},{"arena":"models","category":"dataviz","elo":1128,"win_rate":45.6,"rank":87},{"arena":"models","category":"gamedev","elo":1104,"win_rate":44.7,"rank":88},{"arena":"models","category":"svg","elo":1025,"win_rate":39.1,"rank":80},{"arena":"models","category":"uicomponent","elo":1103,"win_rate":42.6,"rank":88},{"arena":"models","category":"website","elo":1127,"win_rate":45.1,"rank":94}],"artificial_analysis":{"intelligence_index":16.9,"coding_index":43.9,"agentic_index":8}},"reasoning":{"mandatory":false}},{"id":"qwen/qwen3-vl-30b-a3b-instruct","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","created":1759794476,"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-30b-a3b-instruct/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":7.9,"coding_index":null,"agentic_index":null}}},{"id":"z-ai/glm-4.6","canonical_slug":"z-ai/glm-4.6","hugging_face_id":"zai-org/GLM-4.6","name":"Z.ai: GLM 4.6","created":1759235576,"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000002","input_cache_read":"0.0000001"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/z-ai/glm-4.6/endpoints"},"benchmarks":{"design_arena":[{"arena":"agents","category":"androidnative","elo":1099,"win_rate":52.6,"rank":29},{"arena":"agents","category":"fullstack","elo":1028,"win_rate":42.3,"rank":38},{"arena":"agents","category":"godotgamedev","elo":1180,"win_rate":53.2,"rank":17},{"arena":"agents","category":"mobileapps","elo":1093,"win_rate":47.2,"rank":36},{"arena":"models","category":"3d","elo":1134,"win_rate":54.2,"rank":70},{"arena":"models","category":"codecategories","elo":1169,"win_rate":54.3,"rank":73},{"arena":"models","category":"dataviz","elo":1164,"win_rate":52.6,"rank":70},{"arena":"models","category":"gamedev","elo":1158,"win_rate":54.8,"rank":68},{"arena":"models","category":"svg","elo":1108,"win_rate":52.1,"rank":58},{"arena":"models","category":"uicomponent","elo":1151,"win_rate":52.5,"rank":73},{"arena":"models","category":"website","elo":1179,"win_rate":54.3,"rank":75}],"artificial_analysis":{"intelligence_index":18.5,"coding_index":45.8,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"anthropic/claude-sonnet-4.5","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.5","created":1759161676,"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-sonnet-20250929/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1163,"win_rate":51,"rank":60},{"arena":"models","category":"asciiart","elo":1210,"win_rate":56.1,"rank":21},{"arena":"models","category":"codecategories","elo":1187,"win_rate":51.5,"rank":59},{"arena":"models","category":"dataviz","elo":1170,"win_rate":47.4,"rank":67},{"arena":"models","category":"gamedev","elo":1173,"win_rate":51.1,"rank":63},{"arena":"models","category":"svg","elo":1114,"win_rate":52.2,"rank":56},{"arena":"models","category":"uicomponent","elo":1176,"win_rate":49.5,"rank":62},{"arena":"models","category":"website","elo":1193,"win_rate":51.8,"rank":64},{"arena":"agents","category":"fullstack","elo":1049,"win_rate":43.7,"rank":36},{"arena":"agents","category":"mobileapps","elo":1134,"win_rate":52.9,"rank":31},{"arena":"agents","category":"webapps","elo":1053,"win_rate":43.2,"rank":37}],"artificial_analysis":{"intelligence_index":20.7,"coding_index":52.1,"agentic_index":15.8}},"reasoning":{"mandatory":false}},{"id":"qwen/qwen3-vl-235b-a22b-instruct","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","created":1758668687,"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000021","completion":"0.0000019","input_cache_read":"0.0000001"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-235b-a22b-instruct/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":9.9,"coding_index":null,"agentic_index":null}}},{"id":"qwen/qwen3-next-80b-a3b-instruct","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","created":1757612213,"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000011","input_cache_read":"0.00000007"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-instruct-2509/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":9.6,"coding_index":null,"agentic_index":null}}},{"id":"deepseek/deepseek-chat-v3.1","canonical_slug":"deepseek/deepseek-chat-v3.1","hugging_face_id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek: DeepSeek V3.1","created":1755779628,"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3.1/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1082,"win_rate":47.9,"rank":90},{"arena":"models","category":"codecategories","elo":1116,"win_rate":47.8,"rank":93},{"arena":"models","category":"dataviz","elo":1099,"win_rate":46.2,"rank":96},{"arena":"models","category":"gamedev","elo":1091,"win_rate":47.1,"rank":93},{"arena":"models","category":"svg","elo":966,"win_rate":38.2,"rank":89},{"arena":"models","category":"uicomponent","elo":1089,"win_rate":47.1,"rank":93},{"arena":"models","category":"website","elo":1127,"win_rate":48,"rank":95}],"artificial_analysis":{"intelligence_index":13.7,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false}},{"id":"openai/gpt-oss-120b","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000037","completion":"0.00000017"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":906,"win_rate":29.4,"rank":113},{"arena":"models","category":"codecategories","elo":967,"win_rate":33.4,"rank":124},{"arena":"models","category":"dataviz","elo":990,"win_rate":43.6,"rank":112},{"arena":"models","category":"gamedev","elo":999,"win_rate":40.5,"rank":114},{"arena":"models","category":"uicomponent","elo":930,"win_rate":35.7,"rank":117},{"arena":"models","category":"website","elo":972,"win_rate":32.5,"rank":127}],"artificial_analysis":{"intelligence_index":11.6,"coding_index":30.4,"agentic_index":3.7}},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-oss-120b:batch","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b (batch)","created":1754414231,"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000000296","completion":"0.000000136"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":906,"win_rate":29.4,"rank":113},{"arena":"models","category":"codecategories","elo":967,"win_rate":33.4,"rank":124},{"arena":"models","category":"dataviz","elo":990,"win_rate":43.6,"rank":112},{"arena":"models","category":"gamedev","elo":999,"win_rate":40.5,"rank":114},{"arena":"models","category":"uicomponent","elo":930,"win_rate":35.7,"rank":117},{"arena":"models","category":"website","elo":972,"win_rate":32.5,"rank":127}],"artificial_analysis":{"intelligence_index":11.6,"coding_index":30.4,"agentic_index":3.7}},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-oss-20b","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"dataviz","elo":935,"win_rate":39.7,"rank":118},{"arena":"models","category":"website","elo":857,"win_rate":27.9,"rank":134}],"artificial_analysis":{"intelligence_index":10,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"openai/gpt-oss-20b:batch","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b (batch)","created":1754414229,"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000024","completion":"0.000000112"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"dataviz","elo":935,"win_rate":39.7,"rank":118},{"arena":"models","category":"website","elo":857,"win_rate":27.9,"rank":134}],"artificial_analysis":{"intelligence_index":10,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true,"supported_efforts":["high","medium","low"],"default_effort":"medium"}},{"id":"qwen/qwen3-coder","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","hugging_face_id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen: Qwen3 Coder 480B A35B","created":1753230546,"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-480b-a35b-07-25/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"codecategories","elo":1147,"win_rate":61.2,"rank":82},{"arena":"models","category":"dataviz","elo":1086,"win_rate":54.9,"rank":99},{"arena":"models","category":"gamedev","elo":1106,"win_rate":58.7,"rank":87},{"arena":"models","category":"uicomponent","elo":1116,"win_rate":61.4,"rank":83},{"arena":"models","category":"website","elo":1163,"win_rate":61.7,"rank":82}],"artificial_analysis":{"intelligence_index":11.9,"coding_index":null,"agentic_index":null}}},{"id":"qwen/qwen3-235b-a22b-2507","canonical_slug":"qwen/qwen3-235b-a22b-07-25","hugging_face_id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","created":1753119555,"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000055"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-07-25/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":999,"win_rate":41.1,"rank":105},{"arena":"models","category":"codecategories","elo":1042,"win_rate":42.6,"rank":109},{"arena":"models","category":"dataviz","elo":1073,"win_rate":49,"rank":102},{"arena":"models","category":"gamedev","elo":957,"win_rate":35,"rank":121},{"arena":"models","category":"uicomponent","elo":968,"win_rate":38.4,"rank":111},{"arena":"models","category":"website","elo":1062,"win_rate":43.6,"rank":111}],"artificial_analysis":{"intelligence_index":12,"coding_index":null,"agentic_index":null}}},{"id":"mistralai/mistral-small-3.2-24b-instruct","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","hugging_face_id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral: Mistral Small 3.2 24B","created":1750443016,"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000009375","completion":"0.00000025"},"top_provider":{"context_length":256000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-3.2-24b-instruct-2506/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"codecategories","elo":911,"win_rate":39.8,"rank":127},{"arena":"models","category":"dataviz","elo":931,"win_rate":43.3,"rank":119},{"arena":"models","category":"gamedev","elo":894,"win_rate":39.4,"rank":128},{"arena":"models","category":"uicomponent","elo":910,"win_rate":40.5,"rank":120},{"arena":"models","category":"website","elo":900,"win_rate":38.3,"rank":132}]}},{"id":"google/gemini-2.5-pro","canonical_slug":"google/gemini-2.5-pro","hugging_face_id":"","name":"Google: Gemini 2.5 Pro","created":1750169544,"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","image":"0.00000125","audio":"0.00000125","input_audio_cache":"0.000000125","web_search":"0.014","internal_reasoning":"0.00001","input_cache_read":"0.000000125","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000015","audio":"0.0000025","input_audio_cache":"0.00000025","input_cache_read":"0.00000025"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":"2026-10-20","links":{"details":"/api/v1/models/google/gemini-2.5-pro/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1086,"win_rate":50.6,"rank":88},{"arena":"models","category":"codecategories","elo":1156,"win_rate":57.5,"rank":76},{"arena":"models","category":"dataviz","elo":1225,"win_rate":68.2,"rank":45},{"arena":"models","category":"gamedev","elo":1117,"win_rate":54.2,"rank":85},{"arena":"models","category":"uicomponent","elo":1143,"win_rate":57.5,"rank":76},{"arena":"models","category":"website","elo":1170,"win_rate":58.4,"rank":77}],"artificial_analysis":{"intelligence_index":16.1,"coding_index":33.3,"agentic_index":1.6}},"reasoning":{"mandatory":true}},{"id":"deepseek/deepseek-r1-0528","canonical_slug":"deepseek/deepseek-r1-0528","hugging_face_id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek: R1 0528","created":1748455170,"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-r1-0528/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1116,"win_rate":53.3,"rank":79},{"arena":"models","category":"codecategories","elo":1142,"win_rate":52.6,"rank":85},{"arena":"models","category":"dataviz","elo":1185,"win_rate":60.9,"rank":58},{"arena":"models","category":"gamedev","elo":1103,"win_rate":49.3,"rank":89},{"arena":"models","category":"svg","elo":1038,"win_rate":48.7,"rank":75},{"arena":"models","category":"uicomponent","elo":1106,"win_rate":54.9,"rank":86},{"arena":"models","category":"website","elo":1154,"win_rate":52.7,"rank":85}],"artificial_analysis":{"intelligence_index":13.1,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":true}},{"id":"meta-llama/llama-guard-4-12b","canonical_slug":"meta-llama/llama-guard-4-12b","hugging_face_id":"meta-llama/Llama-Guard-4-12B","name":"Meta: Llama Guard 4 12B","created":1745975193,"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":163840,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000018","completion":"0.00000018"},"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-guard-4-12b/endpoints"}},{"id":"qwen/qwen3-30b-a3b","canonical_slug":"qwen/qwen3-30b-a3b-04-28","hugging_face_id":"Qwen/Qwen3-30B-A3B","name":"Qwen: Qwen3 30B A3B","created":1745878604,"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000012","completion":"0.0000005"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-04-28/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"codecategories","elo":946,"win_rate":37.5,"rank":125},{"arena":"models","category":"dataviz","elo":969,"win_rate":39,"rank":114},{"arena":"models","category":"gamedev","elo":906,"win_rate":33.9,"rank":127},{"arena":"models","category":"uicomponent","elo":949,"win_rate":42.4,"rank":114},{"arena":"models","category":"website","elo":959,"win_rate":37.7,"rank":129}],"artificial_analysis":{"intelligence_index":7.6,"coding_index":null,"agentic_index":null}},"reasoning":{"mandatory":false,"default_enabled":true}},{"id":"qwen/qwen3-14b","canonical_slug":"qwen/qwen3-14b-04-28","hugging_face_id":"Qwen/Qwen3-14B","name":"Qwen: Qwen3 14B","created":1745876478,"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":40960,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000012","completion":"0.00000024"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-14b-04-28/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":8.2,"coding_index":13.8,"agentic_index":0.9}},"reasoning":{"mandatory":false}},{"id":"qwen/qwen3-32b","canonical_slug":"qwen/qwen3-32b-04-28","hugging_face_id":"Qwen/Qwen3-32B","name":"Qwen: Qwen3 32B","created":1745875945,"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen3-32b-04-28/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":8.6,"coding_index":15.3,"agentic_index":0.9}},"reasoning":{"mandatory":false}},{"id":"meta-llama/llama-4-maverick","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","hugging_face_id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","name":"Meta: Llama 4 Maverick","created":1743881822,"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":"0.0000001875","completion":"0.0000006525","input_cache_read":"0.00000005"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-4-maverick-17b-128e-instruct/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":906,"win_rate":40.2,"rank":114},{"arena":"models","category":"codecategories","elo":883,"win_rate":35.8,"rank":128},{"arena":"models","category":"dataviz","elo":884,"win_rate":38.4,"rank":124},{"arena":"models","category":"gamedev","elo":845,"win_rate":33.7,"rank":130},{"arena":"models","category":"uicomponent","elo":903,"win_rate":40.8,"rank":121},{"arena":"models","category":"website","elo":875,"win_rate":34.4,"rank":133}],"artificial_analysis":{"intelligence_index":10,"coding_index":16.3,"agentic_index":0.6}}},{"id":"meta-llama/llama-4-scout","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","hugging_face_id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","name":"Meta: Llama 4 Scout","created":1743881519,"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":1310720,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-4-scout-17b-16e-instruct/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"codecategories","elo":793,"win_rate":26.6,"rank":131},{"arena":"models","category":"dataviz","elo":898,"win_rate":39.3,"rank":122},{"arena":"models","category":"gamedev","elo":780,"win_rate":27.4,"rank":131},{"arena":"models","category":"uicomponent","elo":771,"win_rate":25.5,"rank":125},{"arena":"models","category":"website","elo":755,"win_rate":22.7,"rank":139}],"artificial_analysis":{"intelligence_index":8.1,"coding_index":8.2,"agentic_index":0.5}}},{"id":"google/gemma-3-4b-it","canonical_slug":"google/gemma-3-4b-it","hugging_face_id":"google/gemma-3-4b-it","name":"Google: Gemma 3 4B","created":1741905510,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000005","completion":"0.0000001"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-4b-it/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":4.8,"coding_index":2.7,"agentic_index":null}}},{"id":"google/gemma-3-12b-it","canonical_slug":"google/gemma-3-12b-it","hugging_face_id":"google/gemma-3-12b-it","name":"Google: Gemma 3 12B","created":1741902625,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000005","completion":"0.00000015"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-12b-it/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":3.8,"coding_index":5.8,"agentic_index":0.1}}},{"id":"google/gemma-3-27b-it","canonical_slug":"google/gemma-3-27b-it","hugging_face_id":"google/gemma-3-27b-it","name":"Google: Gemma 3 27B","created":1741756359,"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000008","completion":"0.00000045","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/api/v1/models/google/gemma-3-27b-it/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":4.9,"coding_index":10.1,"agentic_index":0.1}}},{"id":"mistralai/mistral-small-24b-instruct-2501","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","hugging_face_id":"mistralai/Mistral-Small-24B-Instruct-2501","name":"Mistral: Mistral Small 3","created":1738255409,"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.00000008"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-24b-instruct-2501/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":6.7,"coding_index":null,"agentic_index":null}}},{"id":"microsoft/phi-4","canonical_slug":"microsoft/phi-4","hugging_face_id":"microsoft/phi-4","name":"Microsoft: Phi 4","created":1736489872,"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.00000014"},"top_provider":{"context_length":16384,"max_completion_tokens":14745,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/microsoft/phi-4/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":5.9,"coding_index":null,"agentic_index":null}}},{"id":"deepseek/deepseek-chat","canonical_slug":"deepseek/deepseek-chat-v3","hugging_face_id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek: DeepSeek V3","created":1735241320,"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.00000089"},"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3/endpoints"},"benchmarks":{"design_arena":[{"arena":"models","category":"3d","elo":1092,"win_rate":50.5,"rank":84},{"arena":"models","category":"codecategories","elo":1112,"win_rate":48.4,"rank":95},{"arena":"models","category":"dataviz","elo":1092,"win_rate":50,"rank":97},{"arena":"models","category":"gamedev","elo":1059,"win_rate":43.7,"rank":102},{"arena":"models","category":"svg","elo":975,"win_rate":38.8,"rank":88},{"arena":"models","category":"uicomponent","elo":1097,"win_rate":52.7,"rank":90},{"arena":"models","category":"website","elo":1124,"win_rate":48.5,"rank":96}]}},{"id":"meta-llama/llama-3.3-70b-instruct","canonical_slug":"meta-llama/llama-3.3-70b-instruct","hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","name":"Meta: Llama 3.3 70B Instruct","created":1733506137,"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000022","completion":"0.0000005","input_cache_read":"0.00000011"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.3-70b-instruct/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":7.7,"coding_index":11.9,"agentic_index":null}}},{"id":"qwen/qwen-2.5-72b-instruct","canonical_slug":"qwen/qwen-2.5-72b-instruct","hugging_face_id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen2.5 72B Instruct","created":1726704000,"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000036","completion":"0.0000004"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/api/v1/models/qwen/qwen-2.5-72b-instruct/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":7.7,"coding_index":null,"agentic_index":null}}},{"id":"sao10k/l3.1-euryale-70b","canonical_slug":"sao10k/l3.1-euryale-70b","hugging_face_id":"Sao10K/L3.1-70B-Euryale-v2.2","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","created":1724803200,"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000085","completion":"0.00000085"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3.1-euryale-70b/endpoints"}},{"id":"nousresearch/hermes-3-llama-3.1-70b","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-70B","name":"Nous: Hermes 3 70B Instruct","created":1723939200,"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.0000007","completion":"0.0000007"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-70b/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":6.7,"coding_index":null,"agentic_index":null}}},{"id":"nousresearch/hermes-3-llama-3.1-405b","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-405B","name":"Nous: Hermes 3 405B Instruct","created":1723766400,"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.000001","completion":"0.000001"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-405b/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":7.4,"coding_index":null,"agentic_index":null}}},{"id":"sao10k/l3-lunaris-8b","canonical_slug":"sao10k/l3-lunaris-8b","hugging_face_id":"Sao10K/L3-8B-Lunaris-v1","name":"Sao10K: Llama 3 8B Lunaris","created":1723507200,"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/sao10k/l3-lunaris-8b/endpoints"}},{"id":"meta-llama/llama-3.1-70b-instruct","canonical_slug":"meta-llama/llama-3.1-70b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-70B-Instruct","name":"Meta: Llama 3.1 70B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.1-70b-instruct/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":6.6,"coding_index":null,"agentic_index":null}}},{"id":"meta-llama/llama-3.1-8b-instruct","canonical_slug":"meta-llama/llama-3.1-8b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Meta: Llama 3.1 8B Instruct","created":1721692800,"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000005","completion":"0.00000008","input_cache_read":"0.000000025"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/api/v1/models/meta-llama/llama-3.1-8b-instruct/endpoints"},"benchmarks":{"design_arena":[],"artificial_analysis":{"intelligence_index":6.9,"coding_index":5.4,"agentic_index":null}}},{"id":"mistralai/mistral-nemo","canonical_slug":"mistralai/mistral-nemo","hugging_face_id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral: Mistral Nemo","created":1721347200,"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000000029","completion":"0.00000003"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/api/v1/models/mistralai/mistral-nemo/endpoints"}},{"id":"gryphe/mythomax-l2-13b","canonical_slug":"gryphe/mythomax-l2-13b","hugging_face_id":"Gryphe/MythoMax-L2-13b","name":"MythoMax 13B","created":1688256000,"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.00000008","completion":"0.00000011"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/api/v1/models/gryphe/mythomax-l2-13b/endpoints"}}],"total_count":114,"links":{"next":null}}