{"object":"list","data":[{"id":"qwen/qwen3.8-27b","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["Qwen/Qwen3.8-27B"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":64,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"vlm-run","task":"chat","task_capabilities":[],"output_tps":null,"context_length":null,"max_output_tokens":null,"name":"Qwen3.8 27B","description":"Reasoning-capable multimodal chat model with a 256k context window and up to 64 images or one video per request.","hf_model_id":"Inferact/Qwen3.8-27B-NVFP4","pricing":{"prompt":0.35,"completion":2.55,"input_cache_read":0.05,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"google/gemini-3.5-flash-lite","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["gemini-3.5-flash-lite"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":true,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url","document_url","file_url"]},"provider":"google-vertex","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":1048576,"max_output_tokens":null,"name":"Gemini 3.5 Flash Lite","description":"Fastest and cheapest Gemini tier. Multimodal chat; emits no reasoning tokens.","hf_model_id":null,"pricing":{"prompt":0.3,"completion":2.5,"input_cache_read":0.03,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"google/gemini-3.7-flash","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["gemini-3.7-flash"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":true,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url","document_url","file_url"]},"provider":"google-vertex","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":1048576,"max_output_tokens":null,"name":"Gemini 3.7 Flash","description":"Newest Gemini Flash tier with extended reasoning. Multimodal chat + native video.","hf_model_id":null,"pricing":{"prompt":0.75,"completion":3.75,"input_cache_read":0.075,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"meta/muse-spark-1.2","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"meta","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":1048576,"max_output_tokens":null,"name":"Muse Spark 1.2","description":"Meta Muse Spark 1.2. Text, multi-image, and native video. Reasoning model.","hf_model_id":null,"pricing":{"prompt":1.25,"completion":4.25,"input_cache_read":0.15,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"google/gemini-robotics-er-2-preview","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["gemini-robotics-er-2-preview","gemini-robotics-er-2"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":true,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url","document_url","file_url"]},"provider":"google-gemini","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":1048576,"max_output_tokens":null,"name":"Gemini Robotics-ER 2 (preview)","description":"Embodied-reasoning VLM for robotics. Text, multi-image, and native video.","hf_model_id":null,"pricing":{"prompt":0.3,"completion":2.5,"input_cache_read":0.03,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"google/gemma-4-26b-a4b-it","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["google/gemma-4-26B-A4B-it","gemma-4-26b-a4b-it"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"openrouter","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":131072,"max_output_tokens":null,"name":"Gemma 4 26B-A4B Instruct","description":"Google Gemma 4 26B MoE (4B active), instruction-tuned. Text, multi-image, and native video.","hf_model_id":"google/gemma-4-26B-A4B-it","pricing":{"prompt":0.1,"completion":0.3,"input_cache_read":0.05,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"moonshotai/kimi-k3","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["moonshotai/Kimi-K3","kimi-k3"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"fireworks","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":262144,"max_output_tokens":null,"name":"Kimi K3","description":"Moonshot Kimi K3. Text, multi-image, and native video.","hf_model_id":"moonshotai/Kimi-K3","pricing":{"prompt":3.0,"completion":15.0,"input_cache_read":0.3,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"minimax/minimax-m3","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["minimax-m3"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"fireworks","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":1048576,"max_output_tokens":null,"name":"MiniMax M3","description":"MiniMax M3. Text, multi-image, and native video. Reasoning model.","hf_model_id":null,"pricing":{"prompt":0.3,"completion":1.2,"input_cache_read":0.06,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"meta/muse-glimmer-30b","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["meta-models/Muse-Glimmer-30B","muse-glimmer-30b"],"methods":[],"default_method":"","extra_body_help":"","capabilities":{"max_images":null,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":false,"supported_input_types":["text","image_url","video_url"]},"provider":"fireworks","task":"chat","task_capabilities":["chat"],"output_tps":null,"context_length":131072,"max_output_tokens":null,"name":"Muse Glimmer 30B","description":"Meta Muse Glimmer 30B. Text, multi-image, and native video.","hf_model_id":"meta-models/Muse-Glimmer-30B","pricing":{"prompt":0.35,"completion":1.5,"input_cache_read":0.04,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"microsoft/florence-2-base-ft","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["microsoft/Florence-2-base-ft"],"methods":["caption","detailed_caption","more_detailed_caption","ocr","ocr_with_region","od","dense_region_caption","region_proposal"],"default_method":"caption","extra_body_help":"{\"method\":\"ocr\"} or {\"method\":\"caption\"}","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":false,"supports_json_schema":false,"supported_input_types":["text","image_url"]},"provider":"","task":"chat","task_capabilities":["ocr","detection","caption"],"output_tps":null,"context_length":null,"max_output_tokens":null,"name":"Florence-2 Base FT","description":"Vision foundation model for captioning, OCR, detection, and region tasks.","hf_model_id":"microsoft/Florence-2-base-ft","pricing":{"prompt":0.1,"completion":0.3,"input_cache_read":0.03,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"geopavlakos/hamer","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":["pose"],"default_method":"pose","extra_body_help":"{\"method\":\"pose\",\"method_params\":{\"body_conf\":0.5,\"hand_conf\":0.5,\"rescale_factor\":2.0,\"include_mesh\":false}}","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":false,"supports_json_schema":false,"supported_input_types":["text","image_url"]},"provider":"","task":"chat","task_capabilities":["pose"],"output_tps":null,"context_length":null,"max_output_tokens":null,"name":"HaMeR","description":"Transformer-based 3D hand mesh recovery with ViTPose-H detection.","hf_model_id":"geopavlakos/HaMeR","pricing":{"prompt":3.9,"completion":0.0,"input_cache_read":3.9,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"nvidia/parakeet-tdt-0.6b-v3","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":["transcribe"],"default_method":"transcribe","extra_body_help":"","capabilities":{"max_images":null,"max_videos":null,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"","task":"transcribe","task_capabilities":["ocr"],"output_tps":null,"context_length":null,"max_output_tokens":null,"name":"Parakeet TDT 0.6B v3","description":"NeMo Parakeet speech-to-text transcription.","hf_model_id":"nvidia/parakeet-tdt-0.6b-v3","pricing":{"prompt":16.67,"completion":0.0,"input_cache_read":16.67,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"paddleocr/pp-ocrv6","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["pp-ocrv6"],"methods":["ocr","detect","text"],"default_method":"ocr","extra_body_help":"{\"method\":\"ocr\"} | {\"method\":\"text\"} | {\"method\":\"ocr\",\"method_params\":{\"lang\":\"en\",\"score_threshold\":0.5}} | document_url PDF (plain text per page)","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":true,"supports_json_schema":false,"supported_input_types":["image_url","document_url","file_url"]},"provider":"","task":"chat","task_capabilities":["ocr"],"output_tps":536.0,"context_length":null,"max_output_tokens":null,"name":"PP-OCRv6","description":"PaddleOCR PP-OCRv6 medium text detection and recognition; scene OCR JSONL on image chat; text returns plain text; document_url fan-out via text.","hf_model_id":null,"pricing":{"prompt":0.01,"completion":0.2,"input_cache_read":0.01,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"facebook/sam3.1","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":["segment","segment_box","track"],"default_method":"segment","extra_body_help":"{\"method\":\"segment\",\"method_params\":{\"prompt\":\"cat\",\"mask_format\":\"png\",\"polygons\":true}} | {\"method\":\"segment_box\",\"method_params\":{\"bbox_xywh\":[0.1,0.1,0.4,0.5]}} | {\"method\":\"track\",\"method_params\":{\"prompt\":\"cat\",\"video_fps\":1,\"video_max_frames\":128}} (Object Multiplex video tracker)","capabilities":{"max_images":1,"max_videos":1,"supports_text_only":false,"supports_document_url":false,"supports_json_schema":false,"supported_input_types":["text","image_url","video_url"]},"provider":"","task":"chat","task_capabilities":["segmentation"],"output_tps":null,"context_length":null,"max_output_tokens":null,"name":"SAM 3.1","description":"SAM 3.1 Object Multiplex: promptable segmentation and faster multi-object video tracking.","hf_model_id":"facebook/sam3.1","pricing":{"prompt":0.0,"completion":0.0,"input_cache_read":0.0,"input_cache_write":0.0,"image":1000.0,"video":34800.0},"is_ready":true},{"id":"usyd-community/vitpose-plus-large","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":["pose"],"default_method":"pose","extra_body_help":"{\"method\":\"pose\"} | video_url (tracked pose every frame; video_fps = detector cadence, default 10)","capabilities":{"max_images":1,"max_videos":1,"supports_text_only":false,"supports_document_url":false,"supports_json_schema":false,"supported_input_types":["text","image_url","video_url"]},"provider":"","task":"chat","task_capabilities":["pose"],"output_tps":null,"context_length":null,"max_output_tokens":null,"name":"ViTPose Plus Large","description":"2D human pose estimation (ViT-L, 434M).","hf_model_id":"usyd-community/vitpose-plus-large","pricing":{"prompt":3.9,"completion":0.0,"input_cache_read":3.9,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"deepseek-ai/deepseek-ocr-2","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["deepseek-ai/DeepSeek-OCR-2","deepseek-ocr-2"],"methods":["ocr","markdown","free_ocr","grounding_ocr"],"default_method":"markdown","extra_body_help":"{\"method\":\"ocr\"} | {\"method\":\"markdown\"} | {\"method\":\"grounding_ocr\",\"method_params\":{\"prompt\":\"<image>\\nLocate <|ref|>title<|/ref|> in the image.\"}} | document_url PDF (markdown per page)","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":true,"supports_json_schema":false,"supported_input_types":["text","image_url","document_url","file_url"]},"provider":"","task":"chat","task_capabilities":["markdown","ocr"],"output_tps":null,"context_length":32768,"max_output_tokens":8192,"name":"DeepSeek OCR 2","description":"DeepSeek OCR 2 and markdown extraction.","hf_model_id":"deepseek-ai/DeepSeek-OCR-2","pricing":{"prompt":0.25,"completion":0.8,"input_cache_read":0.25,"input_cache_write":0.0,"image":30.0,"video":0.0},"is_ready":true},{"id":"rednote-hilab/dots.mocr","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["dots.mocr"],"methods":["parse_layout","parse_layout_only","ocr","markdown"],"default_method":"markdown","extra_body_help":"{\"method\":\"parse_layout\"} | {\"method\":\"markdown\"} | {\"method\":\"ocr\"} | document_url PDF (markdown per page)","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":true,"supports_json_schema":false,"supported_input_types":["text","image_url","document_url","file_url"]},"provider":"","task":"chat","task_capabilities":["markdown"],"output_tps":null,"context_length":32768,"max_output_tokens":16384,"name":"dots.mocr","description":"Multilingual document layout parsing and markdown OCR.","hf_model_id":"rednote-hilab/dots.mocr","pricing":{"prompt":0.2,"completion":0.4,"input_cache_read":0.03,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"zai-org/glm-ocr","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["glm-ocr","zai-org/GLM-OCR"],"methods":["markdown"],"default_method":"markdown","extra_body_help":"{\"method\":\"markdown\"} | document_url PDF (markdown per page)","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":true,"supports_json_schema":false,"supported_input_types":["text","image_url","document_url","file_url"]},"provider":"","task":"chat","task_capabilities":["markdown"],"output_tps":601.0,"context_length":8192,"max_output_tokens":8192,"name":"GLM-OCR","description":"Compact multilingual document OCR and markdown extraction.","hf_model_id":"zai-org/GLM-OCR","pricing":{"prompt":0.1,"completion":0.2,"input_cache_read":0.02,"input_cache_write":0.0,"image":30.0,"video":0.0},"is_ready":true},{"id":"paddlepaddle/paddleocr-vl-1.6","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["PaddlePaddle/PaddleOCR-VL","PaddlePaddle/PaddleOCR-VL-1.6","paddleocr-vl","paddleocr-vl-1.6","paddlepaddle/paddleocr-vl"],"methods":["ocr","table","formula","chart"],"default_method":"ocr","extra_body_help":"{\"method\":\"ocr\"} | {\"method\":\"table\"} | {\"method\":\"formula\"} | {\"method\":\"chart\"} | {\"method\":\"markdown\"} | document_url PDF (markdown per page; uses OCR prompt)","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":true,"supports_json_schema":false,"supported_input_types":["text","image_url","document_url","file_url"]},"provider":"","task":"chat","task_capabilities":["ocr"],"output_tps":null,"context_length":16384,"max_output_tokens":null,"name":"PaddleOCR-VL 1.6","description":"PaddleOCR-VL-1.6 for OCR, tables, formulas, and charts.","hf_model_id":"PaddlePaddle/PaddleOCR-VL-1.6","pricing":{"prompt":0.15,"completion":0.35,"input_cache_read":0.15,"input_cache_write":0.0,"image":30.0,"video":0.0},"is_ready":true},{"id":"qwen/qwen3.5-0.8b","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":[],"default_method":"","extra_body_help":"Multimodal chat (text + up to 64 images or 1 base64 video). No document_url — replies pass through verbatim.","capabilities":{"max_images":64,"max_videos":1,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"","task":"chat","task_capabilities":["chat","caption","detection"],"output_tps":44.0,"context_length":262144,"max_output_tokens":null,"name":"Qwen3.5 0.8B","description":"Multimodal chat model with up to 64 images or one video per request.","hf_model_id":"Qwen/Qwen3.5-0.8B","pricing":{"prompt":0.08,"completion":0.15,"input_cache_read":0.02,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true},{"id":"qwen/qwen3-vl-embedding-2b","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":[],"methods":["embed"],"default_method":"embed","extra_body_help":"","capabilities":{"max_images":null,"max_videos":null,"supports_text_only":true,"supports_document_url":false,"supports_json_schema":true,"supported_input_types":["text","image_url","video_url"]},"provider":"","task":"embed","task_capabilities":[],"output_tps":null,"context_length":8192,"max_output_tokens":null,"name":"Qwen3-VL Embedding 2B","description":"Multimodal embedding model for text, image, and video inputs.","hf_model_id":"Qwen/Qwen3-VL-Embedding-2B","pricing":{"prompt":0.013,"completion":0.0,"input_cache_read":0.013,"input_cache_write":0.0,"image":0.225,"video":0.0},"is_ready":true},{"id":"baidu/unlimited-ocr","object":"model","created":1789512078,"owned_by":"vlm-run","aliases":["baidu/Unlimited-OCR","unlimited-ocr"],"methods":["markdown","multi_page"],"default_method":"markdown","extra_body_help":"{\"method\":\"markdown\"} | {\"method\":\"multi_page\",\"method_params\":{\"window_size\":8}} | document_url PDF (markdown per page; multi_page reads an 8-page sliding window per call)","capabilities":{"max_images":1,"max_videos":0,"supports_text_only":false,"supports_document_url":true,"supports_json_schema":false,"supported_input_types":["text","image_url","document_url","file_url"]},"provider":"","task":"chat","task_capabilities":["markdown","ocr"],"output_tps":null,"context_length":32768,"max_output_tokens":24576,"name":"Unlimited-OCR","description":"Long-horizon document parsing with batched multi-page OCR.","hf_model_id":"baidu/Unlimited-OCR","pricing":{"prompt":0.25,"completion":0.55,"input_cache_read":0.03,"input_cache_write":0.0,"image":0.0,"video":0.0},"is_ready":true}]}