{"data":[{"id":"qwen3.7-plus","name":"Qwen3.7-Plus","object":"model","owned_by":"qwen","info":{"id":"qwen3.7-plus","user_id":"4a8cd1c0-a7f6-4cfa-b5dd-d6bdab408271","base_model_id":null,"name":"Qwen3.7-Plus","meta":{"profile_image_url":"/static/favicon.png","description":"Qwen3.7-Plus is a high-performance large language model within the Qwen3.7 family, integrating state-of-the-art text and multimodal processing capabilities. It can autonomously invoke tools during everyday conversations and excels in web development, artifacts, complex reasoning, role-playing, creative writing, visual reasoning, OCR, and spatial understanding.","capabilities":{"vision":true,"document":true,"video":true,"audio":true,"thinking":true,"search":true},"short_description":"The high-performance large language model in the Qwen3.7 series, supporting text and multimodal tasks.","max_context_length":1000000,"max_summary_generation_length":65536,"abilities":{"vision":1,"document":1,"video":1,"audio":1,"mcp":1,"thinking":3,"parse_url":2},"auto_thinking":true,"auto_search":true,"thinking_format":"summary","chat_type":["t2t","t2v","t2i","image_edit","search","artifacts","web_dev","deep_research","travel","learn","slides","vqa","translate"],"mcp":["image-generation","code-interpreter","amap","fire-crawl"],"modality":["text","image","video"],"think_skip":{"enable":true}},"access_control":null,"is_active":true,"is_visitor_active":true,"updated_at":1732711466,"created_at":1732711466},"preset":true,"action_ids":[]},{"id":"qwen3.8-max","name":"Qwen3.8-Max","object":"model","owned_by":"qwen","info":{"id":"qwen3.8-max","user_id":"4a8cd1c0-a7f6-4cfa-b5dd-d6bdab408271","base_model_id":null,"name":"Qwen3.8-Max","meta":{"profile_image_url":"/static/favicon.png","description":"Qwen3.8-Max is the flagship model of the Qwen3.8 series, delivering state-of-the-art performance across both language and vision modalities. It excels in expert-level knowledge, complex logical reasoning, advanced mathematics, and sophisticated coding tasks. Its vision-language capabilities enable high-precision image understanding, visual reasoning, OCR, document and chart analysis, and fine-grained visual grounding, providing a unified multimodal experience.","capabilities":{"vision":true,"document":true,"video":true,"audio":true,"thinking":true,"search":true},"short_description":"The flagship of Qwen3.8 model delivering state-of-the-art performance.","max_context_length":1000000,"max_summary_generation_length":131072,"abilities":{"vision":1,"document":1,"video":1,"audio":1,"mcp":1,"thinking":3,"parse_url":2},"auto_thinking":true,"auto_search":true,"thinking_format":"summary","chat_type":["t2t","t2v","t2i","image_edit","search","artifacts","web_dev","deep_research","travel","learn","slides","vqa","translate"],"mcp":["image-generation","code-interpreter","amap","fire-crawl"],"modality":["text","image","video"],"think_skip":{"enable":true}},"access_control":null,"is_active":true,"is_visitor_active":true,"updated_at":1732711466,"created_at":1732711466},"preset":true,"action_ids":[]},{"id":"qwen3.8-omni-flash","name":"Qwen3.8-Omni-Flash","object":"model","owned_by":"qwen","info":{"id":"qwen3.8-omni-flash","user_id":"4a8cd1c0-a7f6-4cfa-b5dd-d6bdab408271","base_model_id":null,"name":"Qwen3.8-Omni-Flash","meta":{"profile_image_url":"/static/favicon.png","description":"Qwen3.8-Omni-Flash is the latest-generation multimodal large language model in the Qwen3.8 series, supporting text, image, audio, and audiovisual understanding. It features a 1-million-token context window and can process up to 3 hours of audio or 2 hours of audiovisual input in a single turn. The model delivers strong performance in audio- and video-centric agentic applications, including video editing, music video creation, film and television production and commentary, generating illustrated summaries from audio and video, and audiovisual dialogue—all of which require integrated processing of text, images, audio, and video.","capabilities":{"vision":true,"document":true,"citations":true,"video":true,"audio":true,"thinking":true},"short_description":"The latest native multimodal large language model in the Qwen3.8 series, supporting text, images, audio, and audiovisual input.","max_context_length":1000000,"max_generation_length":131076,"abilities":{"vision":1,"document":1,"citations":1,"video":1,"audio":1,"thinking":1},"chat_type":["t2t","t2i","search","vqa","translate"],"modality":["text","image","video","audio"]},"access_control":null,"is_active":true,"is_visitor_active":false,"updated_at":1732711466,"created_at":1732711466},"preset":true,"action_ids":[]}]}