{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen3 VL 32B Instruct","owned_by":"qwen","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_window":131072,"max_tokens":32768,"type":"language","tags":["tool-use","vision"],"released":1761231332,"modalities":{"input":["text","image"],"output":["text"]},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"pricing":{"input":"0.000000104","output":"0.000000416"},"legal":[]}