{"data":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","created":1761231332,"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","architecture":{"tokenizer":"Qwen","instruct_type":null,"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"]},"endpoints":[{"name":"Alibaba | qwen/qwen3-vl-32b-instruct","model_id":"qwen/qwen3-vl-32b-instruct","model_name":"Qwen: Qwen3 VL 32B Instruct","context_length":131072,"pricing":{"prompt":"0.000000104","completion":"0.000000416","discount":0},"provider_name":"Alibaba","tag":"alibaba/fp8","quantization":"fp8","max_completion_tokens":32768,"max_prompt_tokens":129024,"supported_parameters":["max_tokens","temperature","top_p","seed","presence_penalty","response_format","tools","tool_choice","structured_outputs","logprobs","top_logprobs","top_k","frequency_penalty","stop"],"status":0,"uptime_last_30m":100,"uptime_last_5m":100,"uptime_last_1d":100,"supports_implicit_caching":false,"supports_voice_cloning":false,"latency_last_30m":null,"throughput_last_30m":null}]}}