{"data":[{"id":"deepseek/deepseek-v4-flash-vision-exp","provider":"openrouter","name":"DeepSeek: DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","type":"chat","context_length":"1048576","status":"active","deprecated":false,"updated_at":"2026-08-21T11:26:03.000Z","last_seen_at":"2026-09-14T14:01:07.000Z"},{"id":"deepseek/deepseek-v4-flash-vision-exp:batch","provider":"openrouter","name":"DeepSeek: DeepSeek V4 Flash Vision Exp (batch)","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","type":"chat","context_length":"1048576","status":"active","deprecated":false,"updated_at":"2026-08-21T11:26:03.000Z","last_seen_at":"2026-09-14T14:01:07.000Z"},{"id":"deepseek-v4-flash-vision-exp","provider":"deepseek","name":"Deepseek V4 Flash Vision Exp","description":null,"type":"image","context_length":"unknown","status":"unlisted","deprecated":false,"updated_at":"2026-08-21T00:00:00.000Z","last_seen_at":"2026-09-09T23:01:04.000Z"},{"id":"qwen/qwen3.8-27b","provider":"openrouter","name":"Qwen: Qwen3.8 27B","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","type":"chat","context_length":"1000000","status":"active","deprecated":false,"updated_at":"2026-08-14T15:55:10.000Z","last_seen_at":"2026-09-14T14:01:07.000Z"},{"id":"deepseek/deepseek-v4-flash-0731","provider":"openrouter","name":"DeepSeek: DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","type":"chat","context_length":"1310720","status":"active","deprecated":false,"updated_at":"2026-07-31T06:21:48.000Z","last_seen_at":"2026-09-14T10:01:04.000Z"},{"id":"deepseek/deepseek-v4-flash-0731:batch","provider":"openrouter","name":"DeepSeek: DeepSeek V4 Flash 0731 (batch)","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","type":"chat","context_length":"1048576","status":"active","deprecated":false,"updated_at":"2026-07-31T06:21:48.000Z","last_seen_at":"2026-09-14T10:01:04.000Z"},{"id":"qwen/qwen3.7-flash","provider":"openrouter","name":"Qwen: Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","type":"chat","context_length":"1000000","status":"active","deprecated":false,"updated_at":"2026-07-27T22:16:01.000Z","last_seen_at":"2026-09-14T10:01:04.000Z"},{"id":"nvidia/llama-nemotron-rerank-vl-1b-v2:free","provider":"openrouter","name":"NVIDIA: Llama Nemotron Rerank VL 1B V2 (free)","description":"Llama Nemotron Rerank VL 1B V2 is a 1.7B multimodal reranking model from NVIDIA. It evaluates the relevance of document images and text against user queries, designed for vision RAG...","type":"embedding","context_length":"10240","status":"active","deprecated":false,"updated_at":"2026-06-09T20:14:14.000Z","last_seen_at":"2026-09-14T10:01:04.000Z"},{"id":"stepfun/step-3.7-flash","provider":"openrouter","name":"StepFun: Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-05-28T16:17:49.000Z","last_seen_at":"2026-09-14T10:01:04.000Z"},{"id":"perceptron/perceptron-mk1","provider":"openrouter","name":"Perceptron: Perceptron Mk1","description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","type":"chat","context_length":"32768","status":"active","deprecated":false,"updated_at":"2026-05-12T14:43:49.000Z","last_seen_at":"2026-09-14T10:01:04.000Z"},{"id":"z-ai/glm-5v-turbo","provider":"openrouter","name":"Z.ai: GLM 5V Turbo","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","type":"chat","context_length":"202752","status":"active","deprecated":false,"updated_at":"2026-04-01T16:37:38.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"rekaai/reka-edge","provider":"openrouter","name":"Reka Edge","description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","type":"chat","context_length":"16384","status":"active","deprecated":false,"updated_at":"2026-03-20T17:16:05.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-9b","provider":"openrouter","name":"Qwen: Qwen3.5-9B","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-03-10T14:19:56.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-9b:batch","provider":"openrouter","name":"Qwen: Qwen3.5-9B (batch)","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-03-10T14:19:56.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-35b-a3b","provider":"openrouter","name":"Qwen: Qwen3.5-35B-A3B","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-02-25T21:10:22.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-27b","provider":"openrouter","name":"Qwen: Qwen3.5-27B","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-02-25T21:10:10.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-122b-a10b","provider":"openrouter","name":"Qwen: Qwen3.5-122B-A10B","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-02-25T21:09:49.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-flash-02-23","provider":"openrouter","name":"Qwen: Qwen3.5-Flash","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","type":"chat","context_length":"1000000","status":"active","deprecated":false,"updated_at":"2026-02-25T21:09:36.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-plus-02-15","provider":"openrouter","name":"Qwen: Qwen3.5 Plus 2026-02-15","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","type":"chat","context_length":"1000000","status":"active","deprecated":false,"updated_at":"2026-02-16T08:10:16.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3.5-397b-a17b","provider":"openrouter","name":"Qwen: Qwen3.5 397B A17B","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2026-02-16T06:23:38.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"nim/meta/llama-3.2-11b-vision-instruct","provider":"together","name":"Llama 3.2 11b Vision Instruct","description":null,"type":"image","context_length":"16384","status":"active","deprecated":false,"updated_at":"2026-02-10T00:00:00.000Z","last_seen_at":"2026-09-14T10:01:03.000Z"},{"id":"nim/meta/llama-3.2-90b-vision-instruct","provider":"together","name":"Llama 3.2 90b Vision Instruct","description":null,"type":"image","context_length":"16384","status":"active","deprecated":false,"updated_at":"2026-02-10T00:00:00.000Z","last_seen_at":"2026-09-14T10:01:03.000Z"},{"id":"mistralai/ministral-8b-2512","provider":"openrouter","name":"Mistral: Ministral 3 8B 2512","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2025-12-02T13:20:54.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"mistralai/ministral-8b-2512:batch","provider":"openrouter","name":"Mistral: Ministral 3 8B 2512 (batch)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2025-12-02T13:20:54.000Z","last_seen_at":"2026-09-14T14:01:08.000Z"},{"id":"mistralai/ministral-3b-2512","provider":"openrouter","name":"Mistral: Ministral 3 3B 2512","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","type":"chat","context_length":"131072","status":"active","deprecated":false,"updated_at":"2025-12-02T13:19:20.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3-vl-32b-instruct","provider":"openrouter","name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","type":"chat","context_length":"131072","status":"active","deprecated":false,"updated_at":"2025-10-23T14:55:32.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3-vl-8b-instruct","provider":"openrouter","name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2025-10-14T17:35:08.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"qwen/qwen3-vl-235b-a22b-instruct","provider":"openrouter","name":"Qwen: Qwen3 VL 235B A22B Instruct","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","type":"chat","context_length":"262144","status":"active","deprecated":false,"updated_at":"2025-09-23T23:04:47.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"z-ai/glm-4.5v","provider":"openrouter","name":"Z.ai: GLM 4.5V","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","type":"chat","context_length":"65536","status":"active","deprecated":false,"updated_at":"2025-08-11T14:24:48.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"bytedance/ui-tars-1.5-7b","provider":"openrouter","name":"ByteDance: UI-TARS 7B ","description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","type":"chat","context_length":"128000","status":"active","deprecated":false,"updated_at":"2025-07-22T17:24:16.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"google/gemma-3-4b-it","provider":"openrouter","name":"Google: Gemma 3 4B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","type":"chat","context_length":"131072","status":"active","deprecated":false,"updated_at":"2025-03-13T22:38:30.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"google/gemma-3-12b-it","provider":"openrouter","name":"Google: Gemma 3 12B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","type":"chat","context_length":"131072","status":"active","deprecated":false,"updated_at":"2025-03-13T21:50:25.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"google/gemma-3-27b-it","provider":"openrouter","name":"Google: Gemma 3 27B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","type":"chat","context_length":"131072","status":"active","deprecated":false,"updated_at":"2025-03-12T05:12:39.000Z","last_seen_at":"2026-09-14T10:01:05.000Z"},{"id":"meta-llama/Llama-Guard-3-11B-Vision-Turbo","provider":"together","name":"Llama Guard 3 11B Vision Turbo","description":null,"type":"image","context_length":"131072","status":"unlisted","deprecated":false,"updated_at":"2024-09-25T05:34:49.000Z","last_seen_at":"2026-02-26T05:45:14.000Z"},{"id":"openai/gpt-4-turbo","provider":"openrouter","name":"OpenAI: GPT-4 Turbo","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","type":"chat","context_length":"128000","status":"active","deprecated":false,"updated_at":"2024-04-09T00:00:00.000Z","last_seen_at":"2026-09-14T10:01:06.000Z"},{"id":"openai/gpt-4-turbo:batch","provider":"openrouter","name":"OpenAI: GPT-4 Turbo (batch)","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","type":"chat","context_length":"128000","status":"active","deprecated":false,"updated_at":"2024-04-09T00:00:00.000Z","last_seen_at":"2026-09-14T10:01:06.000Z"}],"meta":{"total":36,"page":1,"limit":50,"pages":1}}