{"data":[{"id":"black-forest-labs/flux-video-edit","canonical_slug":"black-forest-labs/flux-video-edit-20260910","hugging_face_id":null,"name":"Black Forest Labs: FLUX Video Edit","created":1789073532,"description":"FLUX Video Edit [fast] takes a source video and an edit prompt and returns a precisely edited video. Add, remove, or replace objects and characters, rebuild the setting, edit on-screen...","context_length":0,"architecture":{"modality":"text+video->video","input_modalities":["text","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/black-forest-labs/flux-video-edit-20260910/endpoints"}},{"id":"minimax/hailuo-3-max","canonical_slug":"minimax/hailuo-3-max-20260901","hugging_face_id":null,"name":"MiniMax: H3 Max","created":1788313310,"description":"MiniMax H3 Max is a video-generation model from MiniMax, jointly released with fal.ai. Derived through additional training from MiniMax H3, it is designed for faster text-to-video and image-to-video generation with...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/hailuo-3-max-20260901/endpoints"}},{"id":"alibaba/wan-3.0-prime","canonical_slug":"alibaba/wan-3.0-prime-20260827","hugging_face_id":null,"name":"Alibaba: Wan 3.0 Prime","created":1787863800,"description":"Wan 3.0 Prime is a fast-mode variant of [Wan 3.0](https://openrouter.ai/alibaba/wan-3.0) from Alibaba. It supports text-to-video and first-frame image-to-video generation.","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/wan-3.0-prime-20260827/endpoints"}},{"id":"alibaba/wan-3.0","canonical_slug":"alibaba/wan-3.0-20260824","hugging_face_id":null,"name":"Alibaba: Wan 3.0","created":1787603856,"description":"Wan 3.0 is a video generation model from Alibaba for text-to-video, image-to-video, and reference-guided video generation. It produces 480p, 720p, or 1080p video with durations from 2 to 30 seconds.","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/wan-3.0-20260824/endpoints"}},{"id":"heygen/avatar-iv","canonical_slug":"heygen/avatar-iv-20260625","hugging_face_id":null,"name":"HeyGen: Avatar IV","created":1787595728,"description":"HeyGen: Avatar IV is an image-to-video model that animates a single photo into an expressive, lip-synced talking-head video. Rather than only matching mouth shapes to words, it interprets the vocal...","context_length":0,"architecture":{"modality":"text+image+audio->video","input_modalities":["text","image","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/heygen/avatar-iv-20260625/endpoints"}},{"id":"black-forest-labs/flux-video-upscale","canonical_slug":"black-forest-labs/flux-video-upscale-20260819","hugging_face_id":null,"name":"Black Forest Labs: FLUX Video Upscale","created":1787173919,"description":"FLUX Video Upscale is a video upscaling model from Black Forest Labs. It enlarges a single source video by 1.5× to 3× while preserving its duration, with an optional prompt...","context_length":0,"architecture":{"modality":"text+video->video","input_modalities":["text","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/black-forest-labs/flux-video-upscale-20260819/endpoints"}},{"id":"bytedance/seedance-2.0-mini","canonical_slug":"bytedance/seedance-2.0-mini-20260811","hugging_face_id":null,"name":"ByteDance: Seedance 2.0 Mini","created":1786552600,"description":"Seedance 2.0 Mini is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video with image, video, and audio inputs. It...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance/seedance-2.0-mini-20260811/endpoints"}},{"id":"bytedance/seedance-2.5","canonical_slug":"bytedance/seedance-2.5-20260807","hugging_face_id":null,"name":"ByteDance: Seedance 2.5","created":1786141253,"description":"Seedance 2.5 is a video generation model from ByteDance. It is suited for long-form storytelling, multimodal reference-based generation, video editing, and video extension. It supports first-frame and first-and-last-frame control, up...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance/seedance-2.5-20260807/endpoints"}},{"id":"black-forest-labs/flux-3-video","canonical_slug":"black-forest-labs/flux-3-video-20260804","hugging_face_id":null,"name":"Black Forest Labs: FLUX.3 Video","created":1785858831,"description":"FLUX.3 Video is a video generation model from Black Forest Labs. It supports text-to-video, image-guided generation with opening and closing keyframes, and video continuation workflows, making it suited for controlled...","context_length":0,"architecture":{"modality":"text+image+video->video","input_modalities":["text","image","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/black-forest-labs/flux-3-video-20260804/endpoints"}},{"id":"minimax/hailuo-3","canonical_slug":"minimax/hailuo-03-20260730","hugging_face_id":null,"name":"MiniMax: H3","created":1785366648,"description":"MiniMax H3 is a lightweight, open-weights video generation model from MiniMax. It is designed for precise multimodal editing and controlled content generation, including instruction-guided edits, text and brand rendering, and...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/hailuo-03-20260730/endpoints"}},{"id":"runway/aleph-2","canonical_slug":"runway/aleph-2-20260729","hugging_face_id":null,"name":"Runway: Aleph 2.0","created":1785339484,"description":"Runway Aleph 2.0 is an in-context video editing model from Runway. It applies text instructions and keyframe-guided edits across existing footage while preserving details that are not meant to change....","context_length":0,"architecture":{"modality":"text+image+video->video","input_modalities":["text","image","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/runway/aleph-2-20260729/endpoints"}},{"id":"runway/gen-4.5","canonical_slug":"runway/gen-4.5-20260729","hugging_face_id":null,"name":"Runway: Gen-4.5","created":1785339483,"description":"Runway Gen-4.5 is a video generation model from Runway for text-to-video and image-to-video workflows. It is designed for cinematic scene creation with strong motion quality, visual fidelity, and prompt adherence....","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/runway/gen-4.5-20260729/endpoints"}},{"id":"x-ai/grok-imagine-video-1.5","canonical_slug":"x-ai/grok-imagine-video-1.5-20260719","hugging_face_id":null,"name":"SpaceXAI: Grok Imagine Video 1.5","created":1784548100,"description":"Grok Imagine Video 1.5 is a video generation model from SpaceXAI. It creates videos from text prompts, with an optional starting image to guide the scene. It can direct subject...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","seed","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-imagine-video-1.5-20260719/endpoints"}},{"id":"alibaba/happyhorse-1.1","canonical_slug":"alibaba/happyhorse-1.1-20260624","hugging_face_id":null,"name":"Alibaba: HappyHorse 1.1","created":1782269643,"description":"HappyHorse 1.1 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/happyhorse-1.1-20260624/endpoints"}},{"id":"alibaba/happyhorse-1.0","canonical_slug":"alibaba/happyhorse-1.0-20260624","hugging_face_id":null,"name":"Alibaba: HappyHorse 1.0","created":1782260324,"description":"HappyHorse 1.0 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/happyhorse-1.0-20260624/endpoints"}},{"id":"x-ai/grok-imagine-video","canonical_slug":"x-ai/grok-imagine-video-20260512","hugging_face_id":null,"name":"SpaceXAI: Grok Imagine Video","created":1779117586,"description":"Grok Imagine Video is SpaceXAI's fast, text-, image-, and reference-conditioned video generation model. It produces short videos (1–15 seconds, 24 fps) at 480p or 720p across seven aspect ratios -...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","seed","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/x-ai/grok-imagine-video-20260512/endpoints"}},{"id":"kwaivgi/kling-v3.0-pro","canonical_slug":"kwaivgi/kling-v3.0-pro-20260429","hugging_face_id":null,"name":"Kling: Video v3.0 Pro","created":1777496206,"description":"Kling v3.0 Pro is Kuaishou's premium video generation model, offering higher visual quality than the Standard tier. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for precise...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/kwaivgi/kling-v3.0-pro-20260429/endpoints"}},{"id":"kwaivgi/kling-v3.0-std","canonical_slug":"kwaivgi/kling-v3.0-std-20260429","hugging_face_id":null,"name":"Kling: Video v3.0 Standard","created":1777496205,"description":"Kling v3.0 Standard is a video generation model from Kuaishou. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for guided scene composition. Clips range from 3 to...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/kwaivgi/kling-v3.0-std-20260429/endpoints"}},{"id":"google/veo-3.1-fast","canonical_slug":"google/veo-3.1-fast-20260320","hugging_face_id":null,"name":"Google: Veo 3.1 Fast","created":1776994666,"description":"Google's mid-tier video generation model balancing speed and quality. Veo 3.1 Fast generates high-quality video from text or image prompts with native synchronized audio, offering faster turnaround than Veo 3.1...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/veo-3.1-fast-20260320/endpoints"}},{"id":"google/veo-3.1-lite","canonical_slug":"google/veo-3.1-lite-20260331","hugging_face_id":null,"name":"Google: Veo 3.1 Lite","created":1776978818,"description":"Google's most cost-effective video generation model, designed for high-volume applications and rapid iteration. Veo 3.1 Lite generates 720p and 1080p video from text or image prompts with native synchronized audio...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/veo-3.1-lite-20260331/endpoints"}},{"id":"kwaivgi/kling-video-o1","canonical_slug":"kwaivgi/kling-video-o1-20260420","hugging_face_id":null,"name":"Kling: Video O1","created":1776704777,"description":"Kling Video O1 is a video generation model from Kuaishou. It supports text and image inputs with video output, enabling text-to-video and image-to-video workflows. It is suited for cinematic content...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/kwaivgi/kling-video-o1-20260420/endpoints"}},{"id":"minimax/hailuo-2.3","canonical_slug":"minimax/hailuo-2.3-20260420","hugging_face_id":null,"name":"MiniMax: Hailuo 2.3","created":1776702740,"description":"Hailuo 2.3 is a video generation model from MiniMax. It accepts text prompts and reference images as input and generates video output, supporting both text-to-video and image-to-video workflows. It is...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/minimax/hailuo-2.3-20260420/endpoints"}},{"id":"alibaba/wan-2.7","canonical_slug":"alibaba/wan-2.7-20260414","hugging_face_id":null,"name":"Alibaba: Wan 2.7","created":1776211362,"description":"Wan 2.7 is a video generation model from Alibaba. It supports text-to-video, image-to-video with first and last frame control, and reference-to-video, where multiple reference images guide the style and content...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/wan-2.7-20260414/endpoints"}},{"id":"bytedance/seedance-2.0","canonical_slug":"bytedance/seedance-2.0-20260414","hugging_face_id":null,"name":"ByteDance: Seedance 2.0","created":1776211362,"description":"Seedance 2.0 is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It is particularly strong at preserving character consistency,...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance/seedance-2.0-20260414/endpoints"}},{"id":"bytedance/seedance-2.0-fast","canonical_slug":"bytedance/seedance-2.0-fast-20260414","hugging_face_id":null,"name":"ByteDance: Seedance 2.0 Fast","created":1776211362,"description":"Seedance 2.0 Fast is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It prioritizes generation speed and lower cost...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance/seedance-2.0-fast-20260414/endpoints"}},{"id":"alibaba/wan-2.6","canonical_slug":"alibaba/wan-2.6-20260327","hugging_face_id":null,"name":"Alibaba: Wan 2.6","created":1774659190,"description":"Alibaba's most advanced video generation model, supporting over 10 visual creation capabilities in a unified system. Wan 2.6 generates 1080p video at 24fps from text, images, reference videos, or audio,...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/alibaba/wan-2.6-20260327/endpoints"}},{"id":"bytedance/seedance-1-5-pro","canonical_slug":"bytedance/seedance-1-5-pro-20260320","hugging_face_id":null,"name":"ByteDance: Seedance 1.5 Pro","created":1774277608,"description":"ByteDance's next-generation audio-visual generation model with a 4.5B parameter Dual-Branch Diffusion Transformer architecture. Seedance 1.5 Pro generates video and audio simultaneously in a single unified pass — eliminating the timing...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/bytedance/seedance-1-5-pro-20260320/endpoints"}},{"id":"openai/sora-2-pro","canonical_slug":"openai/sora-2-pro-20260320","hugging_face_id":null,"name":"OpenAI: Sora 2 Pro","created":1774277521,"description":"OpenAI's flagship video generation model, delivering production-quality video with physics-accurate motion, synchronized audio, and world-state persistence across shots. Sora 2 Pro follows intricate multi-shot instructions while maintaining consistent spatial relationships...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","presence_penalty","stop","top_logprobs"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/openai/sora-2-pro-20260320/endpoints"}},{"id":"google/veo-3.1","canonical_slug":"google/veo-3.1-20260320","hugging_face_id":null,"name":"Google: Veo 3.1","created":1774277148,"description":"Google's state-of-the-art video generation model, built for maximum visual fidelity in final production cuts. Veo 3.1 generates high-quality 1080p video from text or image prompts with native synchronized audio —...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/api/v1/models/google/veo-3.1-20260320/endpoints"}}],"total_count":29,"links":{"next":null}}