{"data":{"id":"z-ai/glm-5.3-flashx","name":"Z.ai: GLM 5.3 FlashX","created":1789744020,"description":"GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture...","architecture":{"tokenizer":"Other","instruct_type":null,"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"]},"endpoints":[{"name":"Z.AI | z-ai/glm-5.3-flashx-20260918","model_id":"z-ai/glm-5.3-flashx","model_name":"Z.ai: GLM 5.3 FlashX","context_length":1048576,"pricing":{"prompt":"0.00000037","completion":"0.00000125","input_cache_read":"0.00000009","discount":0},"provider_name":"Z.AI","tag":"z-ai/fp8","quantization":"fp8","max_completion_tokens":131072,"max_prompt_tokens":null,"supported_parameters":["reasoning","include_reasoning","max_tokens","temperature","top_p","tools","tool_choice","top_k","response_format","reasoning_effort"],"supports_tool_choice":{"none":false,"auto":true,"required":false,"function":false},"status":0,"uptime_last_30m":100,"uptime_last_5m":100,"uptime_last_1d":99.99289996494358,"supports_implicit_caching":false,"supports_voice_cloning":false,"supports_multiple_audio_references":false,"supports_image_reference":false,"latency_last_30m":null,"throughput_last_30m":null}]}}