{"auto_groups":["default"],"data":[{"model_name":"glm-5.1","description":"Zhipu GLM-5.1 — flagship foundation model built for extended autonomous work: the vendor documents continuous unattended execution on a single task for up to 8 hours across planning, execution, testing, and iterative optimization. 200K context window, up to 128K output tokens, text in and text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":200000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,200K","vendor_id":2,"quota_type":0,"model_ratio":0.7,"model_price":0,"owner_by":"","completion_ratio":3.142857142857143,"cache_ratio":0.18571428571428572,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","pricing_version":"5a90f2b86c08bd983a9a2e6d66c255f4eaef9c4bc934386d2b6ae84ef0ff1f1f","billing_unit":"tokens","book_faces":{"standard":{"cache_read":260000,"input":1400000,"output":4400000},"paid":{"cache_read":260000,"input":1120000,"output":3520000},"route_count":3}},{"model_name":"happyhorse-1.1-i2v","description":"Alibaba HappyHorse 1.1 (I2V) — image-to-video with better texture, ID consistency across clips, motion fluidity, and audio-visual sync. 720P/1080P.","login_required":0,"tags":"Video,Image-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.14,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"per_second","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.14,"selling_usd":0.14,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":140000},"paid":null,"route_count":1}},{"model_name":"mimo-v2.5-pro","description":"Xiaomi MiMo-V2.5-Pro — trillion-parameter flagship (42B active) with 1M context and 128K max output, tuned for high-intensity agent workloads. Supports deep thinking, function calling, structured output, and web search. Text generation only: unlike mimo-v2.5, the vendor's capability list does not include full-modality understanding.","login_required":0,"tags":"Reasoning,Tools,Files,1M","vendor_id":16,"quota_type":0,"model_ratio":0.2175,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.008275862068965517,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.435,"list_input_usd":0.435,"selling_usd":0.435,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.435,"selling_usd":0.435,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.87,"selling_usd":0.87,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.0036,"selling_usd":0.0036,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":3600,"input":435000,"output":870000},"paid":null,"route_count":1}},{"model_name":"MiniMax-M3","description":"MiniMax M3 — frontier multimodal coding model for agentic reasoning, tool use, and structured task execution, with a 1,000,000-token context window. Accepts text, images up to 10 MB, and video up to 50 MB inline or 512 MB through the Files API, and returns text. Supports tool calling, with thinking optional rather than always on.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,1M","vendor_id":13,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.3,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.3,"selling_usd":0.3,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.2,"selling_usd":1.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.06,"selling_usd":0.06,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":60000,"input":300000,"output":1200000},"paid":null,"route_count":1,"tier_count":4,"has_request_condition":true,"tiers":[{"tier":"priority_long_context","standard":{"cache_read":180000,"input":900000,"output":3600000},"paid":null,"when":[{"dim":"request_attr","path":"service_tier","equals":"priority"},{"dim":"context_len","context_len_gt":512000}]},{"tier":"priority","standard":{"cache_read":90000,"input":450000,"output":1800000},"paid":null,"when":[{"dim":"request_attr","path":"service_tier","equals":"priority"}]},{"tier":"long_context","standard":{"cache_read":120000,"input":600000,"output":2400000},"paid":null,"when":[{"dim":"context_len","context_len_gt":512000}]},{"tier":"standard","standard":{"cache_read":60000,"input":300000,"output":1200000},"paid":null}]}},{"model_name":"wan2.7-image","description":"Alibaba Wan2.7 Image — text-to-image, image editing, multi-image reference generation, interactive editing, strong text rendering and instruction following.","login_required":0,"tags":"Image","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.03,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":30000},"paid":null,"route_count":1}},{"model_name":"claude-opus-4-5","description":"Anthropic Claude Opus 4.5 — the Opus 4.5 release, which Anthropic now lists among its legacy models and keeps available, for workloads that want to stay on this model version. It supports text and image input, extended thinking, a 200,000-token context window, and up to 64,000 output tokens.","specs":"{\"type\": \"llm\", \"context_window\": 200000, \"max_output_tokens\": 64000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\"], \"pricing_notes\": [\"Standard: $5 input, $0.50 cache hit, $6.25 5-minute cache write, and $25 output per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal","vendor_id":20,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":10000000,"input":5000000,"output":25000000},"paid":{"cache_read":350000,"cache_write":4375000,"cache_write_1h":7000000,"input":3500000,"output":17500000},"route_count":1}},{"model_name":"doubao-seed-evolving","description":"ByteDance Doubao Seed Evolving — Doubao's current coding and agent flagship, published as a rolling release: the vendor updates the model weekly under this same ID with no dated snapshot, so behavior and capabilities can change without a version bump. Carries the vendor's deep thinking, text generation, multimodal understanding, GUI task handling, tool calling, and structured output (json_schema recommended) capability labels. 1,024K context window with 1,024K maximum input and up to 256K output tokens (4K by default).","specs":"{\"type\": \"llm\", \"context_window\": 1024000, \"max_input_tokens\": 1024000, \"max_output_tokens\": 256000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"release_notes\": [\"Rolling release: updated weekly by the vendor under the same model ID.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,1M","vendor_id":5,"quota_type":0,"model_ratio":0.444444,"model_price":0,"owner_by":"","completion_ratio":5.0000045000045,"cache_ratio":0.199999324999325,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":177777,"input":888888,"output":4444444},"paid":null,"route_count":1}},{"model_name":"glm-5.2","description":"Zhipu GLM-5.2 — flagship foundation model specialised for long-horizon coding-agent work through months of dedicated training rather than a context extension alone, with a 1M-token context window and up to 128K output tokens. Text in, text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":2,"quota_type":0,"model_ratio":0.7,"model_price":0,"owner_by":"","completion_ratio":3.142857142857143,"cache_ratio":0.18571428571428572,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":260000,"input":1400000,"output":4400000},"paid":{"cache_read":260000,"input":1120000,"output":3520000},"route_count":3}},{"model_name":"mimo-v2.6-flash","description":"Xiaomi MiMo-V2.6-Flash — efficient, low-cost reasoning model with open weights, natively full-modality: one model understands images, video, audio and text. 1M context and 128K max output, with deep thinking, function calling and structured output. Aimed at high-volume, everyday professional workloads where cost and throughput matter.","login_required":0,"tags":"Reasoning,Tools,Files,Vision,Video,Audio,Multimodal,Open Weights,Low-cost,1M","vendor_id":16,"quota_type":0,"model_ratio":0.07,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic","openai-response"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","reasoning","vision","file-video-understanding","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-09-22","faces":[{"face":"input","list_usd":0.14,"selling_usd":0.14,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.28,"selling_usd":0.28,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.0028,"selling_usd":0.0028000000000000004,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":2800,"input":140000,"output":280000},"paid":null,"route_count":1}},{"model_name":"qwen3.5-livetranslate-flash-realtime","description":"Alibaba Qwen3.5-LiveTranslate-Flash-Realtime — simultaneous audio and video interpretation over a WebSocket session, built on the Qwen3.5-Omni base with cross-language, cross-modal alignment and visual enhancement. Listens in 60 languages and speaks 29. Reached over GET /v1/realtime. Its price faces differ from the other realtime models: there is no text-input rate — input is billed as audio or image, and output as text or audio.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"audio\",\"image\",\"video\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Realtime,Audio,Voice,Multilingual,Fast","vendor_id":4,"quota_type":0,"model_ratio":0.275,"model_price":0,"owner_by":"","completion_ratio":36.36363636363637,"cache_ratio":1,"image_ratio":1,"audio_ratio":13.636363636363637,"audio_completion_ratio":54.54545454545455,"enable_groups":["default"],"supported_endpoint_types":["realtime"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.55,"list_input_usd":0.55,"selling_usd":0.55,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.55,"selling_usd":0.55,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":20,"selling_usd":20.000000000000004,"discount_pct":-0,"verdict":"at_list"}],"unverified_faces":["cache_read","image_input","audio_input","audio_output"]},"book_faces":{"standard":{"audio_input":7500000,"audio_output":30000000,"image_input":550000,"input":550000,"output":20000000},"paid":null,"route_count":1}},{"model_name":"qwen3.8-flash","description":"Alibaba Qwen3.8-Flash — the fast tier of the Qwen3.8 line, a natively multimodal model the vendor positions for coding assistance, agent collaboration, and image/document understanding at high throughput. 1,000,000-token context window. Accepts text, image, and video input and returns text; supports thinking mode, function calling, built-in tools, and structured outputs. Speaks both the OpenAI and the Anthropic Messages protocol.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Vision,1M,Fast","vendor_id":4,"quota_type":0,"model_ratio":0.075,"model_price":0,"owner_by":"","completion_ratio":3.1333333333333333,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.15,"list_input_usd":0.15,"selling_usd":0.15,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.15,"selling_usd":0.15,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.47,"selling_usd":0.47,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":150000,"output":470000},"paid":null,"route_count":1}},{"model_name":"wan2.7-image-pro","description":"Alibaba Wan 2.7 Image Pro — the professional tier of the Wan 2.7 image line, covering text-to-image, sequential image sets, image editing, and multi-image reference generation with improved prompt following. Text-to-image reaches 4K (4096x4096) with 1K and 2K tiers and custom sizes from 768x768; editing and sequential modes cap at 2K. Accepts up to 9 reference images (JPEG, JPG, PNG, BMP, WEBP; 240-8000 px per side, 20 MB each), aspect ratios from 1:8 to 8:1, and prompts up to 5000 characters. Returns PNG.","login_required":0,"tags":"Image,4K","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.075,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":75000},"paid":null,"route_count":1}},{"model_name":"MiniMax-M2.7","description":"MiniMax M2.7 — text model for chat, coding, office work, and agentic tasks, with a 204,800-token context window and output at roughly 60 tokens per second. Text only: the API accepts text and tool-call content blocks, not images or video. Supports tool calling; thinking is always on and cannot be disabled.","specs":"{\"type\":\"llm\",\"context_window\":204800,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.2,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":60000,"cache_write":375000,"input":300000,"output":1200000},"paid":null,"route_count":1}},{"model_name":"agnes-video-2.5","description":"Agnes AI Video 2.5 — a paid asynchronous video model generating 4-12 second clips at 720P, 960P or 2K from text, first/last keyframes, or image, audio and video references.","specs":"{\"type\":\"video\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"video\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Video,Text-to-Video,Image-to-Video","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0.025,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.025,"list_input_usd":0.025,"selling_usd":0.025,"unit":"per_second","list_source":"https://www.agnes-ai.com/zh-Hans/docs/agnes-video-25","verified_at":"2026-09-01","faces":[{"face":"input","list_usd":0.025,"selling_usd":0.025,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":25000},"paid":null,"route_count":1}},{"model_name":"claude-sonnet-5","description":"Anthropic Claude Sonnet 5 — a production model balancing frontier intelligence and speed for coding, agents, and enterprise workflows. It can plan, use browser and terminal tools, and run autonomously across reasoning, knowledge-work, and software-engineering tasks.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"low\",\"medium\",\"high\",\"max\"],\"pricing_notes\":[\"Introductory Standard price through August 31, 2026: $2 input and $10 output per 1M tokens.\",\"From September 1, 2026: $3 input and $15 output per 1M tokens; cache rates change proportionally.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Files","vendor_id":20,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2,"list_input_usd":2,"selling_usd":2,"unit":"text","list_source":"https://platform.claude.com/docs/en/about-claude/pricing","verified_at":"2026-09-06","faces":[{"face":"input","list_usd":2,"selling_usd":2,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":10,"selling_usd":10,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.2,"selling_usd":0.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":2.5,"selling_usd":2.5,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_write_1h"]},"book_faces":{"standard":{"cache_read":200000,"cache_write":2500000,"cache_write_1h":4000000,"input":2000000,"output":10000000},"paid":{"cache_read":140000,"cache_write":1750000,"cache_write_1h":2800000,"input":1400000,"output":7000000},"route_count":5}},{"model_name":"gemini-2.5-pro","description":"Google Gemini 2.5 Pro — the reasoning-tier model of the 2.5 family, with a 1M-token context window. Prompts above 200,000 tokens are repriced at the vendor's long-context rate for the whole request, matching Google's own rule.","specs":"{\"type\": \"llm\", \"context_window\": 1048576, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\", \"video\", \"audio\", \"pdf\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":21,"quota_type":0,"model_ratio":0.625,"model_price":0,"owner_by":"","completion_ratio":8,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":125000,"input":1250000,"output":10000000},"paid":{"cache_read":87500,"input":875000,"output":7000000},"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":250000,"input":2500000,"output":15000000},"paid":{"cache_read":175000,"input":1750000,"output":10500000},"when":[{"dim":"context_len","context_len_gt":200000}]},{"tier":"standard","standard":{"cache_read":125000,"input":1250000,"output":10000000},"paid":{"cache_read":87500,"input":875000,"output":7000000}}]}},{"model_name":"gemini-3-pro-image","description":"Google Gemini 3 Pro Image (Nano Banana Pro) — a premium, reasoning-driven model for professional image generation and conversational editing. It excels at complex graphic design, high-fidelity product mockups, accurate multilingual text rendering, brand consistency, and factual visualizations grounded with Google Search. It accepts text and images, returns text and images, supports thinking, and generates 1K, 2K, or 4K output.","specs":"{\"type\":\"image\",\"context_window\":65536,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\",\"image\"],\"request_protocol\":\"OpenAI Chat Completions\",\"image_response_path\":\"choices[0].message.content (Markdown data:image URL)\",\"reasoning_efforts\":[\"low\",\"high\"],\"resolutions\":[\"1K\",\"2K\",\"4K\"],\"health_probe\":{\"endpoint_type\":\"openai\",\"prompt\":\"Generate a simple blue circle centered on a white background.\"},\"pricing_notes\":[\"Standard: $2.00 input, $12.00 text/thinking output, and $120.00 image output per 1M tokens.\",\"Per image: $0.134 at 1K/2K and $0.24 at 4K.\",\"When routing is enabled, call POST /v1/chat/completions and read the Markdown data:image asset from choices[0].message.content.\"]}","login_required":0,"icon":"Gemini.Color","tags":"Google,image","vendor_id":21,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","image"],"tasks":["chat","text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"image_output":120000000,"input":2000000,"output":12000000},"paid":{"image_output":84000000,"input":1400000,"output":8400000},"route_count":2}},{"model_name":"gpt-6-luna","description":"OpenAI GPT-6 Luna — OpenAI's most efficient model, built for focused, high-volume tasks. It supports text and image input, a 1,050,000-token context window, up to 128,000 output tokens, and reasoning effort from none to max, with medium as the default. Use /v1/responses for tool calling: on /v1/chat/completions, function calling works only with reasoning_effort set to none. On /v1/responses, send the whole conversation in the input array on every turn; requests that carry previous_response_id or conversation are refused.","specs":"{\"type\": \"llm\", \"context_window\": 1050000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\", \"xhigh\", \"max\"], \"pricing_notes\": [\"Standard: $0.10 input, $0.01 cached input, $0.125 cache write, and $0.50 output per 1M tokens.\", \"Above 272K input tokens, the full request uses 2x input and cache rates and 1.5x output pricing.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":19,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"text","list_source":"https://developers.openai.com/api/docs/pricing","verified_at":"2026-09-24","faces":[{"face":"input","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.5,"selling_usd":0.5,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.01,"selling_usd":0.010000000000000002,"discount_pct":-0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.125,"selling_usd":0.125,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":10000,"cache_write":125000,"input":100000,"output":500000},"paid":{"cache_read":7000,"cache_write":87500,"input":70000,"output":350000},"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":20000,"cache_write":250000,"input":200000,"output":750000},"paid":{"cache_read":14000,"cache_write":175000,"input":140000,"output":525000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":10000,"cache_write":125000,"input":100000,"output":500000},"paid":{"cache_read":7000,"cache_write":87500,"input":70000,"output":350000}}]}},{"model_name":"gpt-image-2.5-sunburst","description":"OpenAI GPT Image 2.5 Sunburst — an image generation and editing model from OpenAI's GPT Image 2.5 line. Sunburst and gpt-image-2.5-flare are separate builds, not aliases of one model, sold at the same price; public leaderboards score them separately. Standard token pricing is $5 text input, $1.25 cached input, $8 image input, and $30 image output per 1M tokens.","specs":"{\"type\": \"image\", \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"image\"], \"resolutions\": [\"1024x1024\", \"1536x1024\", \"1024x1536\", \"auto\"], \"health_probe\": {\"endpoint_type\": \"image-generation\", \"prompt\": \"A simple blue circle on a white background.\", \"size\": \"1024x1024\"}, \"pricing_notes\": [\"Standard: $5 text input, $1.25 cached input, $8 image input, and $30 image output per 1M tokens.\"]}","login_required":0,"tags":"OpenAI,image","vendor_id":19,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.25,"image_ratio":1.6,"enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":5,"list_input_usd":5,"selling_usd":5,"unit":"text","list_source":"https://developers.openai.com/api/docs/pricing","verified_at":"2026-09-22","faces":[{"face":"input","list_usd":5,"selling_usd":5,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":30,"selling_usd":30,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":1.25,"selling_usd":1.25,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["image_input","image_output"]},"book_faces":{"standard":{"cache_read":1250000,"image_input":8000000,"image_output":30000000,"input":5000000,"output":30000000},"paid":{"cache_read":1250000,"image_input":6400000,"image_output":24000000,"input":4000000,"output":24000000},"route_count":2}},{"model_name":"kimi-k2.7-code","description":"Moonshot Kimi K2.7 Code — Kimi's dedicated coding model, which the vendor measures as improving instruction compliance and long-horizon coding over K2.6 while cutting overthinking by about 30% on average. 256K context window, 32,768 default output tokens. Accepts text, images, and video as base64 content and returns text. Thinking is mandatory and returns an error if disabled; tool_choice is limited to auto or none and reasoning_content must be carried across multi-step tool calls.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,256K","vendor_id":3,"quota_type":0,"model_ratio":0.475,"model_price":0,"owner_by":"","completion_ratio":4.2105263157894735,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.95,"list_input_usd":0.95,"selling_usd":0.95,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.95,"selling_usd":0.95,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":4,"selling_usd":3.9999999999999996,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.19,"selling_usd":0.19,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":190000,"input":950000,"output":4000000},"paid":{"cache_read":152000,"input":760000,"output":3200000},"route_count":2}},{"model_name":"cogview-4","description":"Zhipu CogView-4 — the fourth-generation CogView text-to-image model, supporting multiple output resolutions and both a base and a high-definition serving tier. Accepts Chinese and English prompts and renders text inside the image. One image per request.","login_required":0,"tags":"Image,Multi-Resolution","vendor_id":2,"quota_type":1,"model_ratio":0,"model_price":0.01,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.01,"list_input_usd":0.01,"selling_usd":0.01,"unit":"per_call","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.01,"selling_usd":0.01,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":10000},"paid":{"per_call":9500},"route_count":1}},{"model_name":"deepseek-v4-pro","description":"DeepSeek V4 Pro 0813 — the higher-capability DeepSeek V4 tier for coding, reasoning, and agentic work, with a 1M-token context window and a 384K maximum output. It runs in both thinking (default) and non-thinking modes and supports tool calling and JSON output. Text in, text out. On /v1/responses, send the whole conversation in the input array on every turn: no conversation state is kept upstream, so requests that carry previous_response_id or conversation are refused.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":384000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":1,"quota_type":0,"model_ratio":0.33,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.03333333333333333,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.66,"list_input_usd":0.66,"selling_usd":0.66,"unit":"text","list_source":"https://api-docs.deepseek.com/quick_start/pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.66,"selling_usd":0.66,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.98,"selling_usd":1.98,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.022,"selling_usd":0.022000000000000002,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":22000,"input":660000,"output":1980000},"paid":null,"route_count":1,"tier_count":2,"has_time_condition":true,"tiers":[{"tier":"peak","standard":{"cache_read":44000,"input":1320000,"output":3960000},"paid":null,"when":[{"dim":"wall_clock","timezone":"Asia/Shanghai","weekdays":[1,2,3,4,5],"hours":[9,10,11,14,15,16,17]}]},{"tier":"standard","standard":{"cache_read":22000,"input":660000,"output":1980000},"paid":null}]}},{"model_name":"LongCat-2.0","description":"Meituan LongCat-2.0 — 1.6T-parameter open MoE model with native 1M context, built for agentic coding and long-horizon tool use. Text-only reasoning model.","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":15,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.3,"unit":"text","list_source":"https://longcat.ai/platform/docs/pricing/long-cat-2.0","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.3,"selling_usd":0.3,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.2,"selling_usd":1.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.006,"selling_usd":0.006,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":6000,"input":300000,"output":1200000},"paid":null,"route_count":1}},{"model_name":"qwen-audio-3.0-tts-plus","description":"Alibaba Qwen-Audio-3.0-TTS-Plus — high-quality speech synthesis on the Qwen-Audio 3.0 base, covering more minority languages and Chinese dialects than the previous generation with markedly more authentic dialect pronunciation. Free-style instruction following and fine-grained tags steer emotion, tone, character, rate, volume, and style; the model is more robust under noise and reverberation. The plus tier favours fidelity and expressiveness, for content production, audiobooks, dubbing, and brand voice. Billed per character (1 token = 1 character). Reachable both as a request/response call and as a realtime WebSocket session.","login_required":0,"tags":"Audio,TTS,Voice,Multilingual,Dialects","vendor_id":4,"quota_type":0,"model_ratio":12.5,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech","realtime"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"longanlingxin","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":25000000,"output":0},"paid":null,"route_count":1}},{"model_name":"qwen3.7-max","description":"Alibaba Qwen3.7-Max — closed-source flagship positioned for programming, office and productivity work, and long-term autonomous execution, with a 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 131,072 output tokens, and a 262,144-token chain-of-thought budget. Text in, text out. Supports function calling, structured outputs, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":131072,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,1M","vendor_id":4,"quota_type":0,"model_ratio":1.25,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.2,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2.5,"list_input_usd":2.5,"selling_usd":2.5,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":2.5,"selling_usd":2.5,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":7.5,"selling_usd":7.5,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.5,"selling_usd":0.5,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":3.125,"selling_usd":3.125,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":500000,"cache_write":3125000,"input":2500000,"output":7500000},"paid":null,"route_count":2}},{"model_name":"stepaudio-3-realtime-preview","description":"StepFun StepAudio 3 Realtime (preview) — StepFun's full-duplex voice model over a WebSocket session at GET /v1/realtime: natural interruption and turn-taking, adaptive reasoning while speaking, and tool calls during the conversation. Free while StepFun's preview runs. StepFun retires the preview name when the free period ends and releases a paid version, so plan for this model ID to stop working.","login_required":0,"tags":"Realtime,Audio,Voice","vendor_id":18,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["realtime"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"audio_input":0,"audio_output":0,"cache_read":0,"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"wan2.7-t2v","description":"Alibaba Wan2.7 (T2V) — text-to-video with upgraded acting, nuanced emotion, intense action, and dramatic camera cuts. Produces 720P or 1080P H.264 MP4 of 2-15 seconds (5s default) in 16:9, 9:16, 1:1, 4:3, or 3:4, from prompts up to 5000 characters. Audio is included by default: with no audio_url the model composes matching background music or sound effects, or a 2-30 second wav or mp3 track can be supplied instead.","login_required":0,"tags":"Video,Text-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.1,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","book_faces":{"standard":{"video_second":100000},"paid":null,"route_count":1}},{"model_name":"wan2.7-videoedit","description":"Alibaba Wan2.7 VideoEdit — natural-language video editing, local or global, reference-image element replacement, motion/effect/camera replication.","login_required":0,"tags":"Video,Video-Edit","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.1,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["video-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","book_faces":{"standard":{"video_second":100000},"paid":null,"route_count":1}},{"model_name":"cogvideox-3","description":"Zhipu CogVideoX-3 — Zhipu's flagship video generation model, covering text-to-video, image-to-video and first-and-last-frame interpolation. Stronger instruction following and physical simulation than the previous generation, with visibly cleaner output and smoother large-amplitude subject motion. Optional AI-generated audio, 5 or 10 second output at 30 or 60 fps, and resolutions from 1280x720 up to 3840x2160. Asynchronous: the submit call returns a task id and the video is fetched once the task reports SUCCESS.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,First-Last-Frame","vendor_id":2,"quota_type":1,"model_ratio":0,"model_price":0.2,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.2,"list_input_usd":0.2,"selling_usd":0.2,"unit":"per_call","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.2,"selling_usd":0.2,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":200000},"paid":{"per_call":160000},"route_count":1}},{"model_name":"doubao-seedance-2-0-fast-260128","description":"ByteDance Seedance 2.0 Fast — cost-efficient video generation carrying the full Seedance 2.0 feature set but capped at 480p and 720p, 24 fps, 4-15 seconds. Covers text-to-video, first-frame and first-and-last-frame image-to-video, multimodal reference generation from 1-9 images and up to 3 reference videos totalling 15 seconds, video editing, video extension, sound-on generation, and a web-search tool; an audio reference has to accompany an image or video. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16. 480p 5s ≈ $0.26.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,Fast","vendor_id":5,"quota_type":0,"model_ratio":2.1,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":4.2,"list_input_usd":4.2,"selling_usd":4.2,"unit":"text","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-09-03","faces":[{"face":"input","list_usd":4.2,"selling_usd":4.2,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":4200000,"output":0},"paid":null,"route_count":1}},{"model_name":"gpt-5.4-mini","description":"OpenAI GPT-5.4 mini — the small GPT-5.4 model for high-volume workloads, coding, computer use, and subagents. It supports text and image input, built-in tools, reasoning effort from none through xhigh, a 400,000-token context window with a 272,000-token maximum input, and up to 128,000 output tokens.","specs":"{\"type\": \"llm\", \"context_window\": 400000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\", \"xhigh\"], \"pricing_notes\": [\"Standard: $0.75 input, $0.075 cached input, and $4.50 output per 1M tokens.\", \"400,000 context window with a 272,000 maximum input; OpenAI publishes no long-context tier for this model.\"]}","login_required":0,"icon":"OpenAI","tags":"Reasoning,Tools,Vision,Multimodal,Fast,Low-cost,Files","vendor_id":19,"quota_type":0,"model_ratio":0.375,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":75000,"input":750000,"output":4500000},"paid":{"cache_read":75000,"input":525000,"output":3150000},"route_count":1}},{"model_name":"kling-v3-omni","description":"Kling 3.0 Omni — reference-based multi-image video generation. Listed price is the std tier (720p, 5s, no audio, no reference video). Requests that do not set mode use the upstream default pro tier, which bills 1.33x — about $0.56 for the same 5s clip; 4k bills 5x. Pass mode=std to be charged the listed price.","login_required":0,"tags":"Video,Reference-to-Video","vendor_id":6,"quota_type":1,"model_ratio":0,"model_price":0.42,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["reference-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":420000},"paid":null,"route_count":1}},{"model_name":"MiniMax-H3","description":"MiniMax H3 (Hailuo 3.0) — next-generation open general-purpose multimodal video model producing native stereo audio in one pass, at 768P or 2K, 4-15 seconds (integers only), 24 fps. Covers text-to-video, image-to-video from a first and/or last frame, and reference generation from images, videos, or audio for character, motion, camera, style, voice, or editing rhythm; image-to-video and reference generation cannot be combined in one request. Accepts H.264/H.265 video, JPG, JPEG, PNG, WEBP, HEIC, and HEIF images, and WAV or MP3 audio, with prompts up to 7000 characters. $0.08/s at 768P, $0.13/s at 2K.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,Open Weights","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.08,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","book_faces":{"standard":{"video_second":80000},"paid":null,"route_count":1}},{"model_name":"MiniMax-M2.7-highspeed","description":"MiniMax M2.7 Highspeed — the same M2.7 model served at roughly 100 tokens per second for low-latency coding and agent workflows, with a 204,800-token context window. Text only: the API accepts text and tool-call content blocks, not images or video. Supports tool calling; thinking is always on and cannot be disabled.","specs":"{\"type\":\"llm\",\"context_window\":204800,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.3,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.1,"create_cache_ratio":0.625,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":60000,"cache_write":375000,"input":600000,"output":2400000},"paid":null,"route_count":1}},{"model_name":"qwen3.6-flash","description":"Alibaba Qwen3.6-Flash — fast vision-language model that the vendor highlights for agentic coding and for spatial intelligence, with marked gains in object localisation and detection. 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 65,536 output tokens, and a 131,072-token chain-of-thought budget. Accepts text, image, and video input and returns text. Supports function calling and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.125,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.2,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.25,"list_input_usd":0.25,"selling_usd":0.25,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.25,"selling_usd":0.25,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.5,"selling_usd":1.5,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.05,"selling_usd":0.05,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.3125,"selling_usd":0.3125,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":50000,"cache_write":312500,"input":250000,"output":1500000},"paid":null,"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":200000,"cache_write":1250000,"input":1000000,"output":4000000},"paid":null,"when":[{"dim":"context_len","context_len_gt":256000}]},{"tier":"standard","standard":{"cache_read":50000,"cache_write":312500,"input":250000,"output":1500000},"paid":null}]}},{"model_name":"happyhorse-1.1-t2v","description":"Alibaba HappyHorse 1.1 (T2V) — text-to-video with improved semantic understanding, camera control, and dynamic generation. Fluid, detailed, physically consistent 720P/1080P video.","login_required":0,"tags":"Video,Text-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.14,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"per_second","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.14,"selling_usd":0.14,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":140000},"paid":null,"route_count":1}},{"model_name":"mimo-v2.5-asr","description":"Xiaomi MiMo v2.5 ASR — speech recognition model optimized for Chinese and English plus Chinese dialects, handling noisy, far-field, and multi-speaker audio as well as lyrics transcription.","login_required":0,"tags":"Audio,ASR,Dialects,Transcription","vendor_id":16,"quota_type":0,"model_ratio":0.616667,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","book_faces":{"standard":{"input":1233334,"output":0},"paid":null,"route_count":1}},{"model_name":"qwen-image-3.0","description":"Alibaba Qwen-Image-3.0 — the standard tier of the Qwen image line, covering text-to-image and image editing, with accurate rendering of small type (down to 10px), 12 languages, and 20+ typefaces. Accepts prompts up to 4,500 tokens, returns up to 6 images per request, and generates up to 2048x2048.","login_required":0,"tags":"Image,Editing","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.03,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.03,"list_input_usd":0.03,"selling_usd":0.03,"unit":"per_call","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.03,"selling_usd":0.03,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":30000},"paid":null,"route_count":1}},{"model_name":"qwen3.6-35b-a3b","description":"Alibaba Qwen3.6-35B-A3B — a 35-billion-parameter native vision-language MoE model with roughly 3 billion active parameters, open-weight, combining linear attention with a sparse mixture of experts for higher inference efficiency. 262,144-token context window, up to 65,536 output tokens, and up to 256 images or 64 videos per request. Accepts text, image, and video input and returns text; supports function calling, built-in tools, and structured outputs.","specs":"{\"type\":\"llm\",\"context_window\":262144,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Open Weights,256K","vendor_id":4,"quota_type":0,"model_ratio":0.1875,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.375,"list_input_usd":0.375,"selling_usd":0.375,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.375,"selling_usd":0.375,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":2.25,"selling_usd":2.25,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":375000,"output":2250000},"paid":null,"route_count":1}},{"model_name":"qwen3.8-2.4t-a95b","description":"Alibaba Qwen3.8-2.4T-A95B — the open-weight release of the Qwen3.8 flagship, a sparse MoE model with 2.4 trillion total parameters and roughly 95 billion active per step, using a hybrid attention design. 1,000,000-token context window. Accepts text input and returns text; thinking and non-thinking modes are billed at the same rate and the output price includes the chain of thought.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":4,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2,"list_input_usd":2,"selling_usd":2,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":2,"selling_usd":2,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":6,"selling_usd":6,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":2000000,"output":6000000},"paid":null,"route_count":1}},{"model_name":"step-5-preview","description":"StepFun step-5-preview — StepFun's new flagship model for coding and professional knowledge work, released as a preview. 1M-token context with up to 64K output tokens, and native understanding of images and video alongside text. Reasoning is always on: it comes back as reasoning_content on /v1/chat/completions and as thinking blocks on /v1/messages, bills as output, and draws from max_tokens, so a limit of a few hundred tokens can be spent entirely on reasoning and return no text; leave generous headroom. Depth is selectable per request through reasoning_effort (low / medium / high). Supports multi-step tool calling, JSON Mode and JSON Schema structured output, streaming, and prompt caching. Built for long-document analysis, software engineering, and multi-step agent work. On /v1/responses, send the whole conversation in the input array on every turn: no conversation state is kept upstream, so requests that carry previous_response_id or conversation are refused.","login_required":0,"tags":"Vision,Video,Reasoning,Tools,1M","vendor_id":18,"quota_type":0,"model_ratio":0.5,"model_price":0,"owner_by":"","completion_ratio":2.7,"cache_ratio":0.05,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1,"list_input_usd":1,"selling_usd":1,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-21","faces":[{"face":"input","list_usd":1,"selling_usd":1,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":2.7,"selling_usd":2.7,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.05,"selling_usd":0.05,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":50000,"input":1000000,"output":2700000},"paid":null,"route_count":1}},{"model_name":"stepaudio-2.5-tts","description":"StepFun StepAudio 2.5 TTS — contextual text-to-speech that carries understanding through the whole generation pipeline instead of treating delivery as post-processing. Voice, emotion, and pacing are directed in natural language at two levels: a global context for the request and inline context for individual passages. Zero-shot voice cloning needs about 3 seconds of reference audio and inherits the full contextual control set. Accepts up to 1000 characters per request; outputs wav, mp3, flac, or opus.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice,VoiceClone","vendor_id":18,"quota_type":0,"model_ratio":42.5,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"cixingnansheng","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":85000000,"output":0},"paid":null,"route_count":1}},{"model_name":"wan3.0-video","description":"Alibaba Wan3.0 (Video) — an all-in-one video model that unifies text-to-video, image-to-video and reference-to-video behind a single endpoint. Generates H.264 MP4 at 480P, 720P or 1080P, up to 30 seconds, from a prompt and an optional first-frame image URL. Billed per second of generated video against the vendor's own ladder: 720P is the base rate, 480P is charged at half of it and 1080P at double. A request that does not name a resolution is produced at 1080P by the model and is charged at that tier. Reference video and reference audio inputs exist in the vendor's model but are not carried on this endpoint.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.1,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"per_second","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-07","faces":[{"face":"input","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":100000},"paid":{"video_second":85000},"route_count":1}},{"model_name":"agnes-image-2.5-flash","description":"Agnes AI Image 2.5 Flash — the latest-generation image model, ahead of Image 2.1 Flash on generation, editing, composition, detail rendering and prompt adherence, for text-to-image, image-to-image, and multi-image composition across flexible 1K–4K sizes and aspect ratios.","specs":"{\"type\": \"image\", \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"image\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Image,Editing,High-definition","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":0},"paid":null,"route_count":1}},{"model_name":"hy4-preview","description":"Tencent Hunyuan 4 preview (Hy4 preview) — 1M-token context window (960K max input, 64K max output) with deep thinking (which Tencent labels preserved thinking), structured output, function calling and prompt caching. Text-only. Preview SKU: Tencent retires a preview model once its general-availability version ships.","login_required":0,"tags":"Reasoning,Tools,1M","vendor_id":17,"quota_type":0,"model_ratio":0.417,"model_price":0,"owner_by":"","completion_ratio":2.998800959232614,"cache_ratio":0.050359712230215826,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.834,"list_input_usd":0.834,"selling_usd":0.834,"unit":"text","list_source":"https://www.tencentcloud.com/document/product/1300/78937","verified_at":"2026-09-27","faces":[{"face":"input","list_usd":0.834,"selling_usd":0.834,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":2.501,"selling_usd":2.501,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.042,"selling_usd":0.041999999999999996,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":42000,"input":834000,"output":2501000},"paid":null,"route_count":1}},{"model_name":"qwen-audio-3.0-tts-flash","description":"Alibaba Qwen-Audio-3.0-TTS-Flash — speech synthesis on the Qwen-Audio 3.0 base tuned for realtime interaction, covering more minority languages and Chinese dialects than the previous generation. Free-style instruction following and fine-grained tags steer emotion, tone, character, rate, and volume; the model is more robust under noise and reverberation. The flash tier optimises for low-latency synthesis, for voice assistants, live dialogue, and support agents. Billed per character (1 token = 1 character). Reachable both as a request/response call and as a realtime WebSocket session.","login_required":0,"tags":"Audio,TTS,Voice,Multilingual,Dialects,Low-latency","vendor_id":4,"quota_type":0,"model_ratio":9,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech","realtime"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"longanhuan_v3.6","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":18000000,"output":0},"paid":null,"route_count":1}},{"model_name":"wan2.7-i2v","description":"Alibaba Wan 2.7 Image-to-Video — covers first-frame generation, first-and-last-frame transitions, and continuation of an existing clip. Produces 720P or 1080P (default 1080P) of 2-15 seconds, 5s default, from prompts up to 5000 characters (negative prompts up to 500 characters). A 2-30 second mp3 or wav driving_audio track can be supplied; with no audio the model generates matching background music or sound effects, audio longer than the clip is truncated, and audio shorter than it leaves the tail silent.","login_required":0,"tags":"Video,Image-to-Video,Audio","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.1,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","book_faces":{"standard":{"video_second":100000},"paid":null,"route_count":1}},{"model_name":"agnes-3.0-flash","description":"Agnes AI Agnes 3.0 Flash — the vendor's new-generation language model, built for agent coding and tool-driven task execution, with a 512K input context and a 65,536-token output limit. Takes text and image-URL input and returns text, and supports streaming, function calling and multi-step tool orchestration, and long-task instruction following. Reachable on the OpenAI-compatible chat endpoint, the Responses endpoint and the Anthropic-compatible Messages endpoint. On the chat endpoint, Thinking mode is switched on by setting chat_template_kwargs.enable_thinking to true; measured 2026-09-21, requests without the field returned no reasoning tokens. With Thinking on, reasoning tokens count toward max_tokens, so size it for the reasoning as well as the answer.","specs":"{\"type\": \"llm\", \"context_window\": 524288, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,512K,Fast","vendor_id":25,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":0,"cache_write":0,"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"doubao-seed-2-1-pro-260628","description":"ByteDance Doubao Seed 2.1 Pro — flagship Doubao agent model for production work, covering deep thinking, text generation, multimodal understanding, tool calling, and structured output, for which the vendor recommends json_schema mode. 256K context window with 256K maximum input, up to 256K output tokens (4K by default), and a 256K chain-of-thought budget. Accepts text, image, and video and returns text; audio understanding is not listed for the 2.1 line. Supports implicit and explicit context caching, including prefix and session caches.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":256000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,256K","vendor_id":5,"quota_type":0,"model_ratio":0.444445,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":888890,"output":4444450},"paid":null,"route_count":1}},{"model_name":"glm-asr-2512","description":"Zhipu GLM-ASR-2512 — speech recognition model reporting a 0.0717 character error rate. Covers Mandarin together with Sichuanese, Cantonese, Min Nan, and Wu, American and British English accents, and dozens of further languages including French, German, Japanese, Korean, Spanish, and Arabic. Handles mixed-language speech, technical jargon, and colloquial delivery, and accepts a custom dictionary for domain vocabulary. Audio is capped at 30 seconds and 25 MB per request, with both batch and streaming transcription.","login_required":0,"tags":"Audio,ASR,Multilingual,Transcription","vendor_id":2,"quota_type":0,"model_ratio":1.2,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","book_faces":{"standard":{"input":2400000,"output":0},"paid":{"input":1920000,"output":0},"route_count":1}},{"model_name":"MiniMax-M2.1","description":"MiniMax M2.1 — text model built for multilingual programming, with a 204,800-token context window and output at roughly 60 tokens per second. Text only: the API accepts text and tool-call content blocks, not images or video. Vendor-designated legacy model that MiniMax states it still serves normally; carried for version compatibility, since customers pin a model name to a generation. Open weights are published on Hugging Face.","specs":"{\"type\": \"llm\", \"context_window\": 204800, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.3,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-09-05","faces":[{"face":"input","list_usd":0.3,"selling_usd":0.3,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.2,"selling_usd":1.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.03,"selling_usd":0.03,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.375,"selling_usd":0.375,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":30000,"cache_write":375000,"input":300000,"output":1200000},"paid":null,"route_count":1}},{"model_name":"minimax-music-v3.0","description":"MiniMax Music 3.0 — synchronous full-song generation from a musical prompt and optional lyrics, with instrumental mode, lyric optimization, and URL or hex audio output. Prompts support up to 2,000 characters and lyrics up to 3,500 characters.","specs":"{\"type\": \"music\", \"input_modalities\": [\"text\"], \"output_modalities\": [\"audio\"], \"pricing_notes\": [\"Standard: CNY 1 per generated song converted at FX 6.75.\"]}","login_required":0,"tags":"Audio,Music,Text-to-Music","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.148148,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["music-generation"],"capabilities":["audio-speech"],"tasks":["music-generation"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":148148},"paid":null,"route_count":1}},{"model_name":"stepaudio-3-chat-preview","description":"StepFun StepAudio 3 Chat (preview) — speech-understanding dialogue model on POST /v1/chat/completions: send speech as input_audio content parts, or text, turn by turn and get text back, with tool calling. Free while StepFun's preview runs. StepFun retires the preview name when the free period ends and releases a paid version, so plan for this model ID to stop working at that point.","login_required":0,"tags":"Audio,Tools","vendor_id":18,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":0,"cache_write":0,"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"agnes-2.5-pro","description":"Agnes AI Agnes 2.5 Pro — the commercial stable release of the 2.5 Pro Alpha benchmark model, a paid reasoning model with a 1M input context and a 65,536-token max output, for advanced coding, scientific reasoning, long-context analysis, agent workflows and image understanding.","specs":"{\"type\": \"llm\", \"context_window\": 1048576, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Reasoning,Tools,Vision,Multimodal,1M","vendor_id":25,"quota_type":0,"model_ratio":0.225,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.1,"create_cache_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.45,"list_input_usd":0.45,"selling_usd":0.45,"unit":"text","list_source":"https://www.agnes-ai.com/zh-Hans/docs/agnes-25-pro","verified_at":"2026-09-01","faces":[{"face":"input","list_usd":0.45,"selling_usd":0.45,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.9,"selling_usd":0.9,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.045,"selling_usd":0.045000000000000005,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":45000,"cache_write":0,"input":450000,"output":900000},"paid":null,"route_count":1}},{"model_name":"claude-opus-4-8","description":"Anthropic Claude Opus 4.8 — a highly capable reasoning model for coding, agentic workflows, professional analysis, and long-running work. It improves reliability, judgment, tool efficiency, computer use, and multimodal document reasoning over Opus 4.7 while supporting a 1,000,000-token context window.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"low\",\"medium\",\"high\",\"max\"],\"pricing_notes\":[\"Standard: $5 input, $25 output, $0.50 cache read, $6.25 5-minute cache write, and $10 1-hour cache write per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":20,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":10000000,"input":5000000,"output":25000000},"paid":{"cache_read":350000,"cache_write":4375000,"cache_write_1h":7000000,"input":3500000,"output":17500000},"route_count":5}},{"model_name":"doubao-seedance-2-5-260628","description":"ByteDance Seedance 2.5 — the mainline Seedance model, stretching a single generation to 4-30 seconds at 24 fps while capping resolution at 480p and 720p. Multimodal reference generation scales to 1-30 images, up to 10 reference videos, and up to 10 reference audio clips (30 seconds total each), and unlike the 2.0 line an audio reference can stand on its own. Also covers text-to-video, first-frame and first-and-last-frame image-to-video, video editing, video extension, sound-on generation, and a web-search tool. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio","vendor_id":5,"quota_type":0,"model_ratio":5.35,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":10700000,"output":0},"paid":null,"route_count":1}},{"model_name":"fun-asr-flash-2026-06-15","description":"Alibaba Fun-ASR Flash (2026-06-15) — low-latency recorded-speech recognition with prompt context and automatic language detection across Chinese dialects and more than 30 languages. Accepts URL, Base64, or uploaded AAC/AMR/AVI/FLAC/FLV/M4A/MKV/MOV/MP3/MP4/MPEG/OGG/OPUS/WAV/WEBM/WMA/WMV audio, up to 5 minutes and 2 GB.","specs":"{\"type\": \"asr\", \"input_modalities\": [\"audio\"], \"output_modalities\": [\"text\"], \"max_duration_seconds\": 300, \"pricing_notes\": [\"Standard: $0.126 per audio hour; 1000 equivalent tokens = one audio minute.\"]}","login_required":0,"tags":"Audio,ASR,Multilingual,Low-latency","vendor_id":4,"quota_type":0,"model_ratio":1.05,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.126,"list_input_usd":0.126,"selling_usd":0.126,"unit":"asr","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-12","faces":[{"face":"input","list_usd":0.126,"selling_usd":0.126,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"input":2100000,"output":0},"paid":null,"route_count":1}},{"model_name":"happyhorse-1.1-r2v","description":"Alibaba HappyHorse 1.1 (R2V) — reference-to-video with up to 9 reference images, strong subject/scene/style consistency and camera control. 720P/1080P.","login_required":0,"tags":"Video,Reference-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.14,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["reference-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"per_second","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.14,"selling_usd":0.14,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":140000},"paid":null,"route_count":1}},{"model_name":"kling-v3","description":"Kuaishou Kling 3.0 — text-to-video and image-to-video with native audio and improved element consistency. Clips run 3-15 seconds (5 by default) in 16:9, 9:16, or 1:1, with the tier chosen through mode: std for 720P, pro for 1080P, and 4k for native 4K. Audio is opt-in through sound=on and defaults to off; image-to-video takes a first frame with an optional tail frame. From $0.083/s (720p).","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Audio,4K","vendor_id":6,"quota_type":1,"model_ratio":0,"model_price":0.42,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":420000},"paid":null,"route_count":1}},{"model_name":"qwen3-tts-flash","description":"Alibaba Qwen3-TTS-Flash — low-latency text-to-speech with 17 expressive built-in voices, covering Chinese, English, German, Italian, Portuguese, Spanish, Japanese, Korean, French, and Russian plus an automatic mode for mixed-language text, and dialect output. Accepts up to 512 tokens of text per request and supports both streaming and non-streaming synthesis.","login_required":0,"tags":"Audio,TTS,Multilingual,Low-latency","vendor_id":4,"quota_type":0,"model_ratio":5.5,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"Cherry","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":11000000,"output":0},"paid":null,"route_count":2}},{"model_name":"step-3.5-flash","description":"StepFun step-3.5-flash — text-only reasoning model with 256K context, aimed at logic, mathematics, software engineering, and deep-research work where reasoning quality has to come with fast, reliable execution. Reasoning depth is selectable per request through reasoning_effort (low / medium / high). Supports tool calling, JSON Schema structured output, streaming, and prompt caching. For image or video input use step-3.7-flash.","login_required":0,"tags":"Reasoning,Tools,256K","vendor_id":18,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.3,"selling_usd":0.30000000000000004,"discount_pct":-0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.02,"selling_usd":0.020000000000000004,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":20000,"input":100000,"output":300000},"paid":null,"route_count":1}},{"model_name":"kling-3.0-turbo","description":"Kuaishou Kling 3.0 Turbo — the value tier of the Kling 3.0 line, generating 720P or 1080P (720P default) at 3-15 seconds (5 by default) in 16:9, 9:16, or 1:1, from a prompt or a first-frame image. Native audio is always included and has no on/off switch. Against kling-v3 it drops the 4K tier, tail-frame control, and video input in exchange for speed and cost. 720p 5s with audio ≈ $0.56.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Audio,Fast","vendor_id":6,"quota_type":1,"model_ratio":0,"model_price":0.56,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":560000},"paid":null,"route_count":1}},{"model_name":"MiniMax-Hailuo-2.3","description":"MiniMax Hailuo 2.3 — text-to-video and image-to-video that the vendor positions on instruction following, at 24 fps, with 768p available as 6-second or 10-second clips and 1080p at 6 seconds. 768p 6s = $0.28.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.28,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":280000},"paid":null,"route_count":1}},{"model_name":"MiniMax-M2","description":"MiniMax M2 — the generation MiniMax first shipped for efficient coding and agent workflows, with a 204,800-token context window. It is the oldest tier the vendor still serves and the only one of the legacy tiers with no highspeed variant. Text only: the API accepts text and tool-call content blocks, not images or video. Vendor-designated legacy model that MiniMax states it still serves normally; carried for version compatibility, since customers pin a model name to a generation. Open weights are published on Hugging Face.","specs":"{\"type\": \"llm\", \"context_window\": 204800, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.3,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-09-05","faces":[{"face":"input","list_usd":0.3,"selling_usd":0.3,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.2,"selling_usd":1.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.03,"selling_usd":0.03,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.375,"selling_usd":0.375,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":30000,"cache_write":375000,"input":300000,"output":1200000},"paid":null,"route_count":1}},{"model_name":"qwen3.8-27b","description":"Alibaba Qwen3.8-27B — the 27-billion-parameter native vision-language dense model of the Qwen3.8 line, open-weight, which the vendor positions above 3.6-27B on coding and office tasks in both the text and the visual modality. 1,000,000-token context window. Accepts text and image input and returns text; thinking and non-thinking modes are billed at the same rate and the output price includes the chain of thought.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Open Weights,1M","vendor_id":4,"quota_type":0,"model_ratio":0.25,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":500000,"output":3000000},"paid":null,"route_count":1}},{"model_name":"stepaudio-3-tts","description":"StepFun StepAudio 3 TTS — text-to-speech from the StepAudio 3 series, built for human-level delivery: timbre, intonation, rhythm and breath follow natural speech, including laughter, hesitation and self-correction, and emotion and tone shift to fit the content. Served on POST /v1/audio/speech and billed per character of input text.","login_required":0,"tags":"Audio,TTS,Voice","vendor_id":18,"quota_type":0,"model_ratio":18,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"cixingnansheng","default_response_format":"mp3","speech_input":"voice-name","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.36,"list_input_usd":0.36,"selling_usd":0.36,"unit":"tts","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-21","faces":[{"face":"input","list_usd":0.36,"selling_usd":0.36,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"input":36000000,"output":0},"paid":null,"route_count":1}},{"model_name":"kimi-k2.7-code-highspeed","description":"Moonshot Kimi K2.7 Code Highspeed — the throughput tier of K2.7 Code, serving the same model at roughly 180 tokens per second and up to 260 in short-context work. 256K context window, 32,768 default output tokens. Accepts text, images, and video as base64 content and returns text. Thinking is mandatory and returns an error if disabled; tool_choice is limited to auto or none and reasoning_content must be carried across multi-step tool calls.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"Moonshot.Color","tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,256K","vendor_id":3,"quota_type":0,"model_ratio":0.95,"model_price":0,"owner_by":"","completion_ratio":4.2105263157894735,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.9,"list_input_usd":1.9,"selling_usd":1.9,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":1.9,"selling_usd":1.9,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":8,"selling_usd":7.999999999999999,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.38,"selling_usd":0.38,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":380000,"input":1900000,"output":8000000},"paid":null,"route_count":1}},{"model_name":"mimo-v2.5-tts","description":"Xiaomi MiMo v2.5 TTS — expressive text-to-speech with eight built-in voices, four Chinese and four English (Mia, Chloe, Milo, Dean), covering Chinese and English plus Northeastern, Sichuanese, Henan, and Cantonese dialect styles. Delivery is directed two ways: natural-language style instructions, and inline audio tags for basic and compound emotions, breathing, sighing, and pacing, including a singing mode unique to this model in the MiMo TTS line. Returns 24 kHz mono wav or pcm16 and supports low-latency streaming, which requires pcm16. 8K context and 8K maximum output.","login_required":0,"tags":"Audio,TTS,Voice","vendor_id":16,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"alloy","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"claude-opus-5-5","description":"Anthropic Claude Opus 5.5 — built for long-running agentic coding and knowledge work, at a lower price than Claude Opus 5. It accepts text and image input, has a 1,000,000-token context window, and supports up to 128,000 output tokens. Adaptive thinking is always on; the effort setting controls how deeply it reasons, with medium as the default.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"low\", \"medium\", \"high\", \"xhigh\", \"max\"], \"pricing_notes\": [\"Standard: $4 input, $20 output, $0.20 cache read, $5 5-minute cache write, and $8 1-hour cache write per 1M tokens.\", \"Adaptive thinking is always on and cannot be disabled; medium is the default effort and the full low-to-max effort ladder is supported.\", \"Forced tool use is not supported: tool_choice any or a named tool returns a 400; auto and none work.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":20,"quota_type":0,"model_ratio":2,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.05,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":4,"list_input_usd":4,"selling_usd":4,"unit":"text","list_source":"https://platform.claude.com/docs/en/about-claude/pricing","verified_at":"2026-09-22","faces":[{"face":"input","list_usd":4,"selling_usd":4,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":20,"selling_usd":20,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.2,"selling_usd":0.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":5,"selling_usd":5,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_write_1h"]},"book_faces":{"standard":{"cache_read":200000,"cache_write":5000000,"cache_write_1h":8000000,"input":4000000,"output":20000000},"paid":{"cache_read":140000,"cache_write":3500000,"cache_write_1h":5600000,"input":2800000,"output":14000000},"route_count":2}},{"model_name":"doubao-seedance-2-0-mini-260615","description":"ByteDance Seedance 2.0 Mini — the lightweight, budget-friendly tier of the Seedance 2.0 line, capped at 480p and 720p, 24 fps, 4-15 seconds. It carries the same documented feature set as the rest of the line: text-to-video, first-frame and first-and-last-frame image-to-video, multimodal reference generation from 1-9 images and up to 3 reference videos totalling 15 seconds, video editing, video extension, sound-on generation, and a web-search tool; an audio reference has to accompany an image or video. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,Fast","vendor_id":5,"quota_type":0,"model_ratio":0.7,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.4,"list_input_usd":1.4,"selling_usd":1.4,"unit":"text","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-09-03","faces":[{"face":"input","list_usd":1.4,"selling_usd":1.4,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":1400000,"output":0},"paid":null,"route_count":1}},{"model_name":"gpt-5.6-terra","description":"OpenAI GPT-5.6 Terra — the balanced GPT-5.6 model for everyday professional work, offering strong intelligence and coding-agent performance at a lower cost than Sol. It supports text and image input, built-in tools, a 1,050,000-token context window, and up to 128,000 output tokens.","specs":"{\"type\":\"llm\",\"context_window\":1050000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"none\",\"low\",\"medium\",\"high\",\"xhigh\",\"max\"],\"pricing_notes\":[\"Standard: $2.50 input, $0.25 cached input, and $15 output per 1M tokens.\",\"Above 272K input tokens, the full request uses 2x input and 1.5x output pricing.\"]}","login_required":0,"icon":"OpenAI","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":19,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":200000,"cache_write":2500000,"cache_write_1h":2000000,"input":2000000,"output":12000000},"paid":{"cache_read":160000,"cache_write":2000000,"cache_write_1h":2000000,"input":1600000,"output":9600000},"route_count":2,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":400000,"cache_write":5000000,"cache_write_1h":4000000,"input":4000000,"output":18000000},"paid":{"cache_read":320000,"cache_write":4000000,"cache_write_1h":4000000,"input":3200000,"output":14400000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":200000,"cache_write":2500000,"cache_write_1h":2000000,"input":2000000,"output":12000000},"paid":{"cache_read":160000,"cache_write":2000000,"cache_write_1h":2000000,"input":1600000,"output":9600000}}]}},{"model_name":"M2-her","description":"MiniMax M2-her — conversational model built for role-play and multi-turn chat, with a 65,536-token context window and replies capped at 2,048 tokens (max_completion_tokens). Beyond system / user / assistant it accepts the vendor's extended message roles on the OpenAI-compatible chat endpoint: user_system (the user's persona), group (the conversation's name) and sample_message_user / sample_message_ai (example exchanges that steer the reply style). Every extended-role message must carry a name field; without it the vendor rejects the whole request with error 2013, while plain system/user/assistant messages need no name. Text only: no image input, and the vendor documents no tool calling or caching. Every call carries roughly 170 tokens of vendor-side persona overhead on top of the visible prompt, billed as input.","specs":"{\"type\": \"llm\", \"context_window\": 65536, \"max_output_tokens\": 2048, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"64K","vendor_id":13,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":300000,"output":1200000},"paid":null,"route_count":1}},{"model_name":"MiniMax-Hailuo-2.3-Fast","description":"MiniMax Hailuo 2.3 Fast — the speed-and-cost tier of Hailuo 2.3, focused on image-to-video and on physical motion realism, at 24 fps, with 768p available as 6-second or 10-second clips and 1080p at 6 seconds. 768p 6s = $0.19.","login_required":0,"tags":"Video,Image-to-Video,Fast","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.19,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":190000},"paid":null,"route_count":1}},{"model_name":"qwen-audio-3.0-realtime-plus","description":"Alibaba Qwen-Audio-3.0-Realtime-Plus — full-duplex speech-to-speech conversation over a WebSocket session, which the vendor reports as first overall in the Artificial Analysis Speech-to-Speech ranking. The standard tier favours answer quality, keeping speech reasoning intact while parallel inference and end-to-end streaming hold response latency low. Reached over GET /v1/realtime. Audio input and audio output are billed on their own faces, well above the text rate.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"text\",\"audio\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Realtime,Audio,Voice,Multilingual","vendor_id":4,"quota_type":0,"model_ratio":0.4,"model_price":0,"owner_by":"","completion_ratio":8,"cache_ratio":1,"audio_ratio":8,"audio_completion_ratio":30,"enable_groups":["default"],"supported_endpoint_types":["realtime"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.8,"list_input_usd":0.8,"selling_usd":0.8,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.8,"selling_usd":0.8,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":6.4,"selling_usd":6.4,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read","audio_input","audio_output"]},"book_faces":{"standard":{"audio_input":6400000,"audio_output":24000000,"input":800000,"output":6400000},"paid":null,"route_count":1}},{"model_name":"step-3.5-flash-2603","description":"StepFun step-3.5-flash-2603 — the agent-tuned 256K snapshot of step-3.5-flash, optimized for token efficiency and a low-inference operating mode. Its reasoning_effort parameter accepts only low and high, where the base model takes three levels, so code that sends medium must be adapted. Text-only; supports tool calling, JSON Schema structured output, streaming, and prompt caching.","login_required":0,"tags":"Reasoning,Tools,256K","vendor_id":18,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.3,"selling_usd":0.30000000000000004,"discount_pct":-0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.02,"selling_usd":0.020000000000000004,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":20000,"input":100000,"output":300000},"paid":null,"route_count":1}},{"model_name":"vidu-video-q3","description":"Vidu Q3 — reference-to-video model with consistent subjects, intelligent scene switching, and synchronized audio/video output. Accepts reference images or named subjects; generates 3–16 seconds at 540P, 720P, or 1080P.","specs":"{\"type\": \"video\", \"min_duration_seconds\": 3, \"max_duration_seconds\": 16, \"resolutions\": [\"540P\", \"720P\", \"1080P\"], \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"video\", \"audio\"]}","login_required":0,"tags":"Video,Reference-to-Video,Audio,Subjects","vendor_id":26,"quota_type":1,"model_ratio":0,"model_price":0.06,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.06,"list_input_usd":0.06,"selling_usd":0.06,"unit":"per_second","list_source":"https://platform.vidu.com/docs/pricing","verified_at":"2026-09-12","faces":[{"face":"input","list_usd":0.06,"selling_usd":0.06,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":60000},"paid":null,"route_count":1}},{"model_name":"claude-sonnet-4-6","description":"Anthropic Claude Sonnet 4.6 — the balanced Sonnet generation, offering a 1,000,000-token context window at standard pricing with up to 128,000 output tokens. It supports text and image input and adaptive thinking, and uses the pre-4.7 tokenizer, so the same text yields roughly 30% fewer tokens than on 4.7-and-later models at the same per-token rate.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\"], \"pricing_notes\": [\"Standard: $3 input, $0.30 cache hit, $3.75 5-minute cache write, and $15 output per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M","vendor_id":20,"quota_type":0,"model_ratio":1.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":300000,"cache_write":3750000,"cache_write_1h":6000000,"input":3000000,"output":15000000},"paid":{"cache_read":210000,"cache_write":2625000,"cache_write_1h":4200000,"input":2100000,"output":10500000},"route_count":5}},{"model_name":"mimo-v2.5","description":"Xiaomi MiMo-V2.5 — natively full-modality model with 1M context and 128K max output, understanding images, video, audio, and text in one model. Supports deep thinking, function calling, structured output, and web search, with full-modality agent capability.","login_required":0,"tags":"Reasoning,Tools,Files,Vision,Video,1M","vendor_id":16,"quota_type":0,"model_ratio":0.07,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.14,"selling_usd":0.14,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.28,"selling_usd":0.28,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.0028,"selling_usd":0.0028000000000000004,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":2800,"input":140000,"output":280000},"paid":null,"route_count":1}},{"model_name":"qwen3.5-omni-plus","description":"Alibaba Qwen3.5-Omni-Plus — the capable tier of the Qwen3.5 omni-modal line, understanding and conversing across text, image, audio, and audio-video. Handles more than 10 hours of audio and more than 400 seconds of 720P (1 FPS) audio-video per conversation, with audio input in 60+ languages and speech output in 30+. 65,536-token context window. Audio input and audio-bearing output are billed on their own faces, above the text rate.","specs":"{\"type\":\"llm\",\"context_window\":65536,\"input_modalities\":[\"text\",\"image\",\"audio\",\"video\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Multimodal,Audio,Vision,Multilingual","vendor_id":4,"quota_type":0,"model_ratio":0.7,"model_price":0,"owner_by":"","completion_ratio":5.928571428571429,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","vision","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.4,"list_input_usd":1.4,"selling_usd":1.4,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":1.4,"selling_usd":1.4,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":8.3,"selling_usd":8.3,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":1400000,"output":8300000},"paid":null,"route_count":1}},{"model_name":"qwen3.7-plus","description":"Alibaba Qwen3.7-Plus — multimodal hybrid agent model that perceives real-world scenes, reads screens and drives GUIs, generates code from visual references, and performs end-to-end navigation inside mobile apps. 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 131,072 output tokens, and a 262,144-token chain-of-thought budget. Accepts text, image, and video input and returns text; supports function calling, structured outputs, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":131072,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.2,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.2,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.4,"list_input_usd":0.4,"selling_usd":0.4,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.4,"selling_usd":0.4,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.6,"selling_usd":1.6,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.08,"selling_usd":0.08000000000000002,"discount_pct":-0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.5,"selling_usd":0.5,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":80000,"cache_write":500000,"input":400000,"output":1600000},"paid":null,"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":240000,"cache_write":1500000,"input":1200000,"output":4800000},"paid":null,"when":[{"dim":"context_len","context_len_gt":256000}]},{"tier":"standard","standard":{"cache_read":80000,"cache_write":500000,"input":400000,"output":1600000},"paid":null}]}},{"model_name":"wan2.7-r2v","description":"Alibaba Wan2.7 (R2V) — reference-to-video, up to 5 mixed image/video refs + audio-timbre reference, stronger character/prop/scene consistency.","login_required":0,"tags":"Video,Reference-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.1,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["reference-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","book_faces":{"standard":{"video_second":100000},"paid":null,"route_count":1}},{"model_name":"agnes-image-2.0-flash","description":"Agnes AI Image 2.0 Flash — a fast image generation and editing model for text-to-image, image-to-image, and multi-image composition, with generated results available as URLs or Base64 data.","specs":"{\"type\":\"image\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"image\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Image,Editing,Fast","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":0},"paid":null,"route_count":1}},{"model_name":"gemini-3.6-flash","description":"Google Gemini 3.6 Flash — a stable multimodal reasoning model that balances speed and intelligence for agentic and real-world tasks, with a 1,048,576-token input context and 65,536-token output limit. It accepts text, images, video, audio, and PDFs, and supports thinking, function calling, code execution, search grounding, structured output, and Computer Use (preview).\n\nPricing note: Google's list price for this model is promotional through December 31, 2026 — $0.75 input / $3.75 output / $0.075 cached input per 1M tokens. From January 1, 2027 the vendor's list price returns to $1.50 / $7.50 / $0.15, and our price follows it at the same parity.","specs":"{\"type\":\"llm\",\"context_window\":1048576,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\",\"audio\",\"pdf\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"medium\",\"high\"]}","login_required":0,"icon":"Gemini.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Files","vendor_id":21,"quota_type":0,"model_ratio":0.375,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","gemini","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"audio_input":750000,"cache_read":75000,"input":750000,"output":3750000},"paid":{"audio_input":750000,"cache_read":52500,"input":525000,"output":2625000},"route_count":2}},{"model_name":"mimo-v2.5-tts-voicedesign","description":"Xiaomi MiMo v2.5 TTS VoiceDesign — voice design model: describe the desired voice in natural language in the instructions field and it synthesises speech in that designed voice.","login_required":0,"tags":"Audio,TTS,VoiceDesign","vendor_id":16,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-design"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_response_format":"wav","speech_input":"voice-design","book_faces":{"standard":{"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"qwen3.5-omni-flash-realtime","description":"Alibaba Qwen3.5-Omni-Flash-Realtime — the fast realtime tier of the Qwen3.5 omni-modal line, held open as a WebSocket session for full-duplex speech conversation. Understands text, image, audio, and audio-video; audio input in 60+ languages and speech output in 30+, with steerable voice, web search, complex function calling, and semantic barge-in. Reached over GET /v1/realtime. Audio input and audio-bearing output are billed on their own faces, well above the text rate.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"text\",\"image\",\"audio\",\"video\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Realtime,Multimodal,Audio,Voice,Multilingual,Fast","vendor_id":4,"quota_type":0,"model_ratio":0.275,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"audio_ratio":8.181818181818182,"audio_completion_ratio":32.18181818181818,"enable_groups":["default"],"supported_endpoint_types":["realtime"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.55,"list_input_usd":0.55,"selling_usd":0.55,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.55,"selling_usd":0.55,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":3.3,"selling_usd":3.3000000000000003,"discount_pct":-0,"verdict":"at_list"}],"unverified_faces":["cache_read","audio_input","audio_output"]},"book_faces":{"standard":{"audio_input":4500000,"audio_output":17700000,"input":550000,"output":3300000},"paid":null,"route_count":1}},{"model_name":"glm-5.3","description":"Zhipu GLM-5.3 — flagship coding-and-agent model post-trained on the same base as GLM-5.2: a 50% gain on Zhipu's internal coding bench, open-source SOTA on Terminal-Bench 3.0 and Agents' Last Exam, and state-of-the-art CyberGym vulnerability-discovery results, with a 1M-token context window and up to 128K output tokens. Text in, text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,1M","vendor_id":2,"quota_type":0,"model_ratio":0.7,"model_price":0,"owner_by":"","completion_ratio":3.142857142857143,"cache_ratio":0.18571428571428572,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":260000,"input":1400000,"output":4400000},"paid":{"cache_read":260000,"input":1330000,"output":4180000},"route_count":3}},{"model_name":"glm-5.3-flash","description":"Zhipu GLM-5.3-Flash — the first natively multimodal model in the GLM-5 line and its low-cost frontier tier: 320B total parameters with 18B active, on a hybrid linear + sparse attention architecture that cuts attention compute and KV cache several-fold against GLM-5.3 while halving both active parameters and layer count against GLM-4.5. Accepts text, images and video; returns text. (The vendor also advertises file input, which is not verified here and therefore not offered.) 1M-token context window and up to 128K output tokens. Supports thinking modes, function calling, structured JSON output, streaming, context caching and MCP tool integration. Positioned for visual coding — front-end, game and 3D work where the model inspects its own rendered output and iterates.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\", \"video\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Video,1M","vendor_id":2,"quota_type":0,"model_ratio":0.075,"model_price":0,"owner_by":"","completion_ratio":3.3333333333333335,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":30000,"input":150000,"output":500000},"paid":{"cache_read":30000,"input":135000,"output":450000},"route_count":1}},{"model_name":"gpt-5.6-sol","description":"OpenAI GPT-5.6 Sol — the flagship GPT-5.6 model for frontier reasoning, complex professional work, coding, science, cybersecurity, and long-horizon agentic workflows. It supports text and image input, built-in tools, a 1,050,000-token context window, up to 128,000 output tokens, and the family’s deepest reasoning settings.","specs":"{\"type\":\"llm\",\"context_window\":1050000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"none\",\"low\",\"medium\",\"high\",\"xhigh\",\"max\"],\"pricing_notes\":[\"Standard: $5 input, $0.50 cached input, and $30 output per 1M tokens.\",\"Above 272K input tokens, the full request uses 2x input and 1.5x output pricing.\"]}","login_required":0,"icon":"OpenAI","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":19,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":5000000,"input":5000000,"output":30000000},"paid":{"cache_read":400000,"cache_write":5000000,"cache_write_1h":5000000,"input":4000000,"output":24000000},"route_count":2,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":1000000,"cache_write":12500000,"cache_write_1h":10000000,"input":10000000,"output":45000000},"paid":{"cache_read":800000,"cache_write":10000000,"cache_write_1h":10000000,"input":8000000,"output":36000000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":5000000,"input":5000000,"output":30000000},"paid":{"cache_read":400000,"cache_write":5000000,"cache_write_1h":5000000,"input":4000000,"output":24000000}}]}},{"model_name":"hy-mt2-pro","description":"Tencent HY-MT2 Pro — flagship machine-translation tier for high-quality multilingual translation through the OpenAI Chat Completions protocol. Up to 4K input tokens and 4K output tokens; text-only.","specs":"{\"type\": \"llm\", \"context_window\": 8192, \"max_input_tokens\": 4096, \"max_output_tokens\": 4096, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Translation,Multilingual,Text","vendor_id":17,"quota_type":0,"model_ratio":0.037037,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":74074,"output":296296},"paid":null,"route_count":1}},{"model_name":"hy3","description":"Tencent Hunyuan 3 (Hy3) — 295B-parameter (21B active) MoE model, native 256K context, three thinking modes (fast / low / deep). Strong at coding agents, long-document understanding, multi-turn context, and multi-step agent workflows. Text-only.","login_required":0,"tags":"Reasoning,Tools,256K","vendor_id":17,"quota_type":0,"model_ratio":0.066,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.132,"list_input_usd":0.132,"selling_usd":0.132,"unit":"text","list_source":"https://www.tencentcloud.com/document/product/1300/78937","verified_at":"2026-09-27","faces":[{"face":"input","list_usd":0.132,"selling_usd":0.132,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.528,"selling_usd":0.528,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.033,"selling_usd":0.033,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":33000,"input":132000,"output":528000},"paid":null,"route_count":1}},{"model_name":"qwen-audio-3.0-asr-flash","description":"Alibaba Qwen-Audio-3.0-ASR-Flash — non-realtime speech recognition on the Qwen-Audio 3.0 foundation, covering 30 languages and the seven major Chinese dialect groups (Mandarin, Wu, Xiang, Gan, Hakka, Min, Yue) plus 20+ regional accents. Punctuation prediction and text normalisation turn numbers, dates, and amounts into standard form; hot words and prompt context raise accuracy on domain vocabulary. Accepts audio up to 5 minutes or 2 GB per request.","login_required":0,"tags":"Audio,ASR,Multilingual,Dialects","vendor_id":4,"quota_type":0,"model_ratio":1.05,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.126,"list_input_usd":0.126,"selling_usd":0.126,"unit":"asr","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.126,"selling_usd":0.126,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"input":2100000,"output":0},"paid":null,"route_count":1}},{"model_name":"glm-5","description":"Zhipu GLM-5 — 744B-parameter MoE with 40B active, trained on 28.5T tokens and using DeepSeek Sparse Attention, with a 200K context window and up to 128K output tokens. Text in, text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":200000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"Zhipu.Color","tags":"Reasoning,Tools,Open Weights,200K","vendor_id":2,"quota_type":0,"model_ratio":0.5,"model_price":0,"owner_by":"","completion_ratio":3.2,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":200000,"input":1000000,"output":3200000},"paid":{"cache_read":200000,"input":950000,"output":3040000},"route_count":1}},{"model_name":"mimo-v2.6-pro-ultraspeed","description":"Xiaomi MiMo-V2.6-Pro UltraSpeed — MiMo-V2.6-Pro served in Xiaomi's ultra-fast mode, with output up to 20x faster than the standard model, for real-time interaction and latency-sensitive production. Same capabilities as mimo-v2.6-pro: full-modality understanding of images, video, audio and text, 1M context, 128K max output, deep thinking, function calling and structured output. Priced at ten times mimo-v2.6-pro; Xiaomi provisions capacity for this mode separately, so heavy bursts can be rate-limited.","login_required":0,"tags":"Reasoning,Tools,Files,Vision,Video,Audio,Multimodal,Fast,Low-latency,1M","vendor_id":16,"quota_type":0,"model_ratio":2.175,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.008275862068965517,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic","openai-response"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","reasoning","vision","file-video-understanding","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":4.35,"list_input_usd":4.35,"selling_usd":4.35,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-09-22","faces":[{"face":"input","list_usd":4.35,"selling_usd":4.35,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":8.7,"selling_usd":8.7,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.036,"selling_usd":0.036,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":36000,"input":4350000,"output":8700000},"paid":null,"route_count":1}},{"model_name":"claude-opus-5","description":"Anthropic Claude Opus 5 — a deep-reasoning model for complex agentic coding and enterprise work, with major gains on long-horizon tasks, code review, document workflows, vision, and multi-agent coordination. It has a 1,000,000-token context window, supports up to 128,000 output tokens, and enables adaptive thinking by default.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"low\",\"medium\",\"high\",\"xhigh\",\"max\"],\"pricing_notes\":[\"Standard: $5 input, $25 output, $0.50 cache read, $6.25 5-minute cache write, and $10 1-hour cache write per 1M tokens.\",\"Adaptive thinking is on by default; high is the default effort and the full low-to-max effort ladder is supported.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":20,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":10000000,"input":5000000,"output":25000000},"paid":{"cache_read":350000,"cache_write":4375000,"cache_write_1h":7000000,"input":3500000,"output":17500000},"route_count":3}},{"model_name":"glm-5-turbo","description":"Zhipu GLM-5-Turbo — the throughput-oriented GLM-5 variant, optimised for agent workflows: more stable tool and skill invocation across multi-step tasks, better handling of complex long-chain instructions, and scheduled or long-running task execution. 200K context window, up to 128K output tokens, text in and text out. Supports thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":200000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,200K","vendor_id":2,"quota_type":0,"model_ratio":0.6,"model_price":0,"owner_by":"","completion_ratio":3.3333333333333335,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":240000,"input":1200000,"output":4000000},"paid":{"cache_read":240000,"input":1140000,"output":3800000},"route_count":1}},{"model_name":"jev-latest","description":"TypeSafe Jev (latest) — the alias the TypeSafe SDKs call when no model is given. It currently serves jev-1.13, at the same price and with the same System One contract on POST /v1/systemone: state plus typed choice, score and noul questions in, typed answers with probabilities and a confidence value out. The response's model field names the version that answered. We move the alias to a new Jev release only after checking its price, so pin jev-1.13 if you have tuned thresholds against it.","specs":"{\"type\": \"decision\", \"context_window\": 32000, \"input_modalities\": [\"text\"], \"output_modalities\": [\"decisions\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/typesafe.png","tags":"Text,Low-latency,Low-cost","vendor_id":27,"quota_type":0,"model_ratio":0.0231,"model_price":0,"owner_by":"","completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["system-one"],"capabilities":["decision"],"tasks":["structured-decision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":46200,"output":0},"paid":null,"route_count":1}},{"model_name":"MiniMax-Hailuo-02","description":"MiniMax Hailuo 02 — budget video generation covering both text-to-video and image-to-video at 24 fps, with 512p and 768p available as 6-second or 10-second clips and 1080p at 6 seconds. The 512p tier is the entry point of the Hailuo line.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.28,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":280000},"paid":null,"route_count":1}},{"model_name":"step-audio-2","description":"StepFun step-audio-2 — end-to-end speech model that consumes audio directly rather than working from a transcript, so tone and delivery survive into the model's understanding. Billed per token, sharing the input rate of the StepAudio 2.5 line at a higher output rate. StepFun publishes no separate capability page for this model; consult the platform audio documentation for the current parameter set.","login_required":0,"tags":"Audio","vendor_id":18,"quota_type":0,"model_ratio":0.7407405,"model_price":0,"owner_by":"","completion_ratio":7,"cache_ratio":0.19999986499995612,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":296296,"input":1481481,"output":10370367},"paid":null,"route_count":1}},{"model_name":"stepaudio-2.5-asr-stream","description":"StepFun StepAudio 2.5 ASR Stream — the streaming-recognition edition of StepAudio 2.5 ASR (a 4B-parameter recognizer built on Multi-Token Prediction, optimized for Chinese and English), and the streaming model StepFun recommends. On this gateway it is called like the other transcription models: upload a complete recording to /v1/audio/transcriptions and the full transcript is returned in one response once recognition finishes; partial results are not streamed back, and it costs more per audio minute than stepaudio-2.5-asr. Inverse text normalization is applied. Accepts 16-bit PCM wav and ogg uploads only (mp3 is not accepted); transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":1.5,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","accepted_upload_formats":["wav","wave","ogg","oga"],"book_faces":{"standard":{"input":3000000,"output":0},"paid":null,"route_count":1}},{"model_name":"wan3.0-video-prime","description":"Alibaba Wan3.0 Video Prime — the fast tier of the Wan 3.0 all-in-one video model, matching the standard tier's capabilities at higher generation speed. Text-to-video, first/last-frame image-to-video and reference-image-to-video (up to 10 reference images) behind one endpoint; H.264 MP4 at 480P, 720P or 1080P, 2 to 30 seconds, 30 fps, with audio by default. Billed per second of generated video on the vendor's ladder: 720P is the base rate, 480P costs just under half of it and 1080P double. A request that names no resolution renders at the vendor's default 1080P and is billed at that tier. Reference video, reference audio, file and web-link inputs are not accepted on this gateway, and smart duration (-1) is rejected; set duration explicitly. Served on a single official Alibaba channel.","specs":"{\"type\": \"video\", \"resolutions\": [\"480P\", \"720P\", \"1080P\"], \"max_duration_seconds\": 30, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"video\"], \"pricing_notes\": [\"Per second of generated video: 480P $0.068, 720P $0.14 (base), 1080P $0.28.\", \"A request without a resolution renders and bills at the vendor's default 1080P tier.\", \"Reference video/audio, file and link inputs are not accepted on this gateway; use first_frame, last_frame and reference_image.\"]}","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Fast","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.14,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"per_second","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-07","faces":[{"face":"input","list_usd":0.14,"selling_usd":0.14,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":140000},"paid":null,"route_count":1}},{"model_name":"claude-opus-4-7","description":"Anthropic Claude Opus 4.7 — an advanced reasoning model for difficult software engineering and complex, long-running agentic tasks. It emphasizes rigorous instruction following, self-verification, reliable tool use, and high-resolution vision across interfaces, images, documents, and professional workflows.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"low\",\"medium\",\"high\"],\"pricing_notes\":[\"Standard: $5 input, $25 output, $0.50 cache read, $6.25 5-minute cache write, and $10 1-hour cache write per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":20,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":10000000,"input":5000000,"output":25000000},"paid":{"cache_read":350000,"cache_write":4375000,"cache_write_1h":7000000,"input":3500000,"output":17500000},"route_count":5}},{"model_name":"doubao-seedream-4-5-251128","description":"ByteDance Doubao Seedream 4.5 — text-to-image and image-to-image with multi-image fusion, producing either a single image or a set of related images. Single-image mode takes a prompt alone, one reference image, or 2-14 reference images; group mode (sequential_image_generation=auto) returns up to 15 images from text, up to 14 from a single reference image, or a set from 2-14 references where inputs plus outputs stay at or below 15. Supports 2K and 4K output and returns JPEG by default, as a URL or base64. Interactive coordinate-based editing is available only in Seedream 5.0 Pro (doubao-seedream-5-0-pro-260628).","login_required":0,"tags":"Image,4K","vendor_id":5,"quota_type":1,"model_ratio":0,"model_price":0.037037,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":37037},"paid":null,"route_count":1}},{"model_name":"gemini-3.1-flash-image","description":"Google Gemini 3.1 Flash Image (Nano Banana 2) — a stable, low-latency model for high-quality image generation and conversational editing. It accepts text, images, and PDFs, supports thinking and search grounding, and can generate 0.5K, 1K, 2K, or 4K images for high-volume production workflows.","specs":"{\"type\":\"image\",\"context_window\":131072,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"pdf\"],\"output_modalities\":[\"text\",\"image\"],\"request_protocol\":\"OpenAI Chat Completions\",\"image_response_path\":\"choices[0].message.content (Markdown data:image URL)\",\"reasoning_efforts\":[\"minimal\",\"high\"],\"resolutions\":[\"0.5K\",\"1K\",\"2K\",\"4K\"],\"health_probe\":{\"endpoint_type\":\"openai\",\"prompt\":\"Generate a simple blue circle centered on a white background.\"},\"pricing_notes\":[\"Standard: $0.50 input, $3.00 text/thinking output, and $60.00 image output per 1M tokens.\",\"Per image: $0.045 at 0.5K, $0.067 at 1K, $0.101 at 2K, and $0.151 at 4K.\",\"Call POST /v1/chat/completions and read the Markdown data:image asset from choices[0].message.content.\"]}","login_required":0,"icon":"Gemini.Color","tags":"Image,Editing,Fast,4K","vendor_id":21,"quota_type":0,"model_ratio":0.25,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","gemini"],"capabilities":["text-multimodal","image"],"tasks":["chat","text-to-image","image-editing","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"image_output":60000000,"input":500000,"output":3000000},"paid":{"image_output":42000000,"input":350000,"output":2100000},"route_count":1}},{"model_name":"gpt-5.5","description":"OpenAI GPT-5.5 — a frontier reasoning model for complex coding and professional work. It supports text and image input, function calling and built-in tools, a 1,050,000-token context window, up to 128,000 output tokens, and reasoning effort from none through xhigh.","specs":"{\"type\":\"llm\",\"context_window\":1050000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"none\",\"low\",\"medium\",\"high\",\"xhigh\"],\"pricing_notes\":[\"Standard: $5 input, $0.50 cached input, and $30 output per 1M tokens.\",\"Above 272K input tokens, the full request uses 2x input and 1.5x output pricing.\"]}","login_required":0,"icon":"OpenAI","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":19,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"input":5000000,"output":30000000},"paid":{"cache_read":400000,"input":4000000,"output":24000000},"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":1000000,"input":10000000,"output":45000000},"paid":{"cache_read":800000,"input":8000000,"output":36000000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":500000,"input":5000000,"output":30000000},"paid":{"cache_read":400000,"input":4000000,"output":24000000}}]}},{"model_name":"qwen3.8-max","description":"Alibaba Qwen3.8-Max — 2.4-trillion-parameter MoE flagship and the most capable model in the Qwen family, with the vendor citing major gains in coding and office productivity. 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 131,072 output tokens, and a 262,144-token chain-of-thought budget. Accepts text, image, and video input and returns text; supports function calling, structured outputs, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":131072,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,1M","vendor_id":4,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.125,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2,"list_input_usd":2,"selling_usd":2,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":2,"selling_usd":2,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":6,"selling_usd":6,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.25,"selling_usd":0.25,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":2.5,"selling_usd":2.5,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":250000,"cache_write":2500000,"input":2000000,"output":6000000},"paid":{"cache_read":237500,"cache_write":2375000,"input":1900000,"output":5700000},"route_count":2}},{"model_name":"vidu-video-q3-pro","description":"Vidu Q3 Pro — quality-oriented text-to-video, image-to-video, and start/end-frame generation with synchronized audio/video output. Generates 1–16 seconds at 540P, 720P, or 1080P; reference-to-video is not supported by this SKU.","specs":"{\"type\": \"video\", \"min_duration_seconds\": 1, \"max_duration_seconds\": 16, \"resolutions\": [\"540P\", \"720P\", \"1080P\"], \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"video\", \"audio\"]}","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,First-Last-Frame,Audio","vendor_id":26,"quota_type":1,"model_ratio":0,"model_price":0.1,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"per_second","list_source":"https://platform.vidu.com/docs/pricing","verified_at":"2026-09-12","faces":[{"face":"input","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":100000},"paid":null,"route_count":1}},{"model_name":"gemini-2.5-flash","description":"Google Gemini 2.5 Flash — a fast, low-cost multimodal model with a 1M-token context window. Accepts text, images, video and audio; audio input is billed at the vendor's separate audio rate.","specs":"{\"type\": \"llm\", \"context_window\": 1048576, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\", \"video\", \"audio\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Tools,Vision,Multimodal,1M,Fast,Low-cost","vendor_id":21,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":8.333333333333334,"cache_ratio":0.1,"audio_ratio":3.3333333333333335,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"audio_input":1000000,"cache_read":30000,"input":300000,"output":2500000},"paid":{"audio_input":1000000,"cache_read":21000,"input":210000,"output":1750000},"route_count":2},"route_health":{"window_hours":24,"sample_size":48,"success_ratio":0.041666666666666664,"last_checked_at":1790602529}},{"model_name":"gemini-3.1-flash-lite-image","description":"Google Gemini 3.1 Flash-Lite Image (Nano Banana 2 Lite) — an ultra-low-latency, cost-efficient model for high-volume image generation and fast multi-turn edits. It accepts text and images, produces 1K images, supports thinking and conversational editing, and is designed for interactive developer and consumer applications.","specs":"{\"type\":\"image\",\"context_window\":65536,\"max_output_tokens\":4096,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\",\"image\"],\"request_protocol\":\"OpenAI Chat Completions\",\"image_response_path\":\"choices[0].message.content (Markdown data:image URL)\",\"reasoning_efforts\":[\"minimal\",\"high\"],\"resolutions\":[\"1024px (1K)\"],\"health_probe\":{\"endpoint_type\":\"openai\",\"prompt\":\"Generate a simple blue circle centered on a white background.\"},\"pricing_notes\":[\"Standard: $0.25 input, $1.50 text/thinking output, and $30.00 image output per 1M tokens.\",\"A 1K image is approximately $0.0336.\",\"Call POST /v1/chat/completions and read the Markdown data:image asset from choices[0].message.content.\"]}","login_required":0,"icon":"Gemini.Color","tags":"Image,Editing,Fast,Low-cost","vendor_id":21,"quota_type":0,"model_ratio":0.125,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","gemini"],"capabilities":["text-multimodal","image"],"tasks":["chat","text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"image_output":30000000,"input":250000,"output":1500000},"paid":{"image_output":21000000,"input":175000,"output":1050000},"route_count":1}},{"model_name":"gpt-5.6-luna","description":"OpenAI GPT-5.6 Luna — the fastest and most affordable GPT-5.6 tier, optimized for cost-sensitive, high-volume workloads while retaining reasoning, vision, and tool use. It supports text and image input, a 1,050,000-token context window, and up to 128,000 output tokens.","specs":"{\"type\":\"llm\",\"context_window\":1050000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"none\",\"low\",\"medium\",\"high\",\"xhigh\",\"max\"],\"pricing_notes\":[\"Standard: $1 input, $0.10 cached input, and $6 output per 1M tokens.\",\"Above 272K input tokens, the full request uses 2x input and 1.5x output pricing.\"]}","login_required":0,"icon":"OpenAI","tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Low-cost,Files","vendor_id":19,"quota_type":0,"model_ratio":0.1,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":20000,"cache_write":250000,"cache_write_1h":200000,"input":200000,"output":1200000},"paid":{"cache_read":17000,"cache_write":212500,"cache_write_1h":200000,"input":170000,"output":1020000},"route_count":2,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":40000,"cache_write":500000,"cache_write_1h":400000,"input":400000,"output":1800000},"paid":{"cache_read":34000,"cache_write":425000,"cache_write_1h":400000,"input":340000,"output":1530000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":20000,"cache_write":250000,"cache_write_1h":200000,"input":200000,"output":1200000},"paid":{"cache_read":17000,"cache_write":212500,"cache_write_1h":200000,"input":170000,"output":1020000}}]}},{"model_name":"qwen3.5-omni-plus-realtime","description":"Alibaba Qwen3.5-Omni-Plus-Realtime — the realtime tier of the Qwen3.5 omni-modal line, held open as a WebSocket session for full-duplex speech conversation. Understands text, image, audio, and audio-video; audio input in 60+ languages and speech output in 30+, with steerable voice, web search, complex function calling, and semantic barge-in. Reached over GET /v1/realtime. Audio input and audio-bearing output are billed on their own faces, well above the text rate.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"text\",\"image\",\"audio\",\"video\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Realtime,Multimodal,Audio,Voice,Multilingual","vendor_id":4,"quota_type":0,"model_ratio":1.05,"model_price":0,"owner_by":"","completion_ratio":5.904761904761905,"cache_ratio":1,"audio_ratio":7.857142857142857,"audio_completion_ratio":29.523809523809526,"enable_groups":["default"],"supported_endpoint_types":["realtime"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2.1,"list_input_usd":2.1,"selling_usd":2.1,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":2.1,"selling_usd":2.1,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":12.4,"selling_usd":12.4,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read","audio_input","audio_output"]},"book_faces":{"standard":{"audio_input":16500000,"audio_output":62000000,"input":2100000,"output":12400000},"paid":null,"route_count":1}},{"model_name":"deepseek-flash","description":"DeepSeek V4.1 Flash — the canonical model id for new integrations. It supports text and image input, thinking and non-thinking modes, tool calling and JSON output. The retired deepseek-v4-flash and deepseek-v4-flash-vision-exp ids remain compatibility aliases served by this same model at the same Flash price. 1M-token context window, 384K maximum output. On /v1/responses, send the whole conversation in the input array on every turn: no conversation state is kept upstream, so requests that carry previous_response_id or conversation are refused.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":384000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,Open Weights,1M","vendor_id":1,"quota_type":0,"model_ratio":0.075,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.15,"list_input_usd":0.15,"selling_usd":0.15,"unit":"text","list_source":"https://api-docs.deepseek.com/quick_start/pricing","verified_at":"2026-09-10","faces":[{"face":"input","list_usd":0.15,"selling_usd":0.15,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.6,"selling_usd":0.6,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.003,"selling_usd":0.003,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":3000,"input":150000,"output":600000},"paid":null,"route_count":1,"tier_count":2,"has_time_condition":true,"tiers":[{"tier":"peak","standard":{"cache_read":6000,"input":300000,"output":1200000},"paid":null,"when":[{"dim":"wall_clock","timezone":"Asia/Shanghai","weekdays":[1,2,3,4,5],"hours":[9,10,11,14,15,16,17]}]},{"tier":"standard","standard":{"cache_read":3000,"input":150000,"output":600000},"paid":null}]}},{"model_name":"doubao-seed-2-1-turbo-260628","description":"ByteDance Doubao Seed 2.1 Turbo — the low-latency lane of the Seed 2.1 line, sharing Pro's envelope: 256K context window with 256K maximum input, up to 256K output tokens (4K by default), and a 256K chain-of-thought budget. Covers deep thinking, text generation, multimodal understanding, tool calling, and structured output, though without Pro's json_schema recommendation. Accepts text, image, and video and returns text; audio understanding is not listed for the 2.1 line. Supports implicit and explicit context caching.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":256000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Fast,256K","vendor_id":5,"quota_type":0,"model_ratio":0.222222,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":444444,"output":2222220},"paid":null,"route_count":1}},{"model_name":"MiniMax-M2.5","description":"MiniMax M2.5 — text model the vendor positions on top-tier performance at a low price for complex tasks, with a 204,800-token context window and output at roughly 60 tokens per second. Text only: the API accepts text and tool-call content blocks, not images or video. Vendor-designated legacy model that MiniMax states it still serves normally; carried for version compatibility, since customers pin a model name to a generation. Open weights are published on Hugging Face.","specs":"{\"type\": \"llm\", \"context_window\": 204800, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.3,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-09-05","faces":[{"face":"input","list_usd":0.3,"selling_usd":0.3,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.2,"selling_usd":1.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.03,"selling_usd":0.03,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.375,"selling_usd":0.375,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":30000,"cache_write":375000,"input":300000,"output":1200000},"paid":null,"route_count":1}},{"model_name":"qwen3.5-omni-flash","description":"Alibaba Qwen3.5-Omni-Flash — the fast tier of the Qwen3.5 omni-modal line, understanding and conversing across text, image, audio, and audio-video. Handles more than 10 hours of audio and more than 400 seconds of 720P (1 FPS) audio-video per conversation, with audio input in 60+ languages and speech output in 30+. Audio input and audio-bearing output are billed on their own faces, above the text rate.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"text\",\"image\",\"audio\",\"video\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Multimodal,Audio,Vision,Multilingual,Fast","vendor_id":4,"quota_type":0,"model_ratio":0.2,"model_price":0,"owner_by":"","completion_ratio":5.5,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","vision","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.4,"list_input_usd":0.4,"selling_usd":0.4,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.4,"selling_usd":0.4,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":2.2,"selling_usd":2.2,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":400000,"output":2200000},"paid":null,"route_count":1}},{"model_name":"claude-fable-5","description":"Anthropic Claude Fable 5 — Anthropic’s most capable generally available model for ambitious, long-running knowledge work and coding. It is built for multi-day agentic tasks, complex software engineering, deep research, document and vision reasoning, and proactive self-verification.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"low\",\"medium\",\"high\",\"max\"],\"pricing_notes\":[\"Standard: $10 input, $50 output, $1 cache read, $12.50 5-minute cache write, and $20 1-hour cache write per 1M tokens.\",\"Adaptive thinking is always enabled.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":20,"quota_type":0,"model_ratio":5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":1000000,"cache_write":12500000,"cache_write_1h":20000000,"input":10000000,"output":50000000},"paid":{"cache_read":700000,"cache_write":8750000,"cache_write_1h":14000000,"input":7000000,"output":35000000},"route_count":4},"route_health":{"window_hours":24,"sample_size":11,"success_ratio":1,"last_checked_at":1790603484}},{"model_name":"gpt-image-2","description":"OpenAI GPT Image 2 — a state-of-the-art image generation and editing model for fast, high-quality output. It takes text and image input, supports flexible image sizes and high-fidelity image inputs, and bills per token rather than per call: $5 text input, $1.25 cached text input, $8 image input, and $30 image output per 1M tokens.","specs":"{\"type\":\"image\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"image\"],\"resolutions\":[\"1024x1024\",\"1536x1024\",\"1024x1536\",\"auto\"],\"health_probe\":{\"endpoint_type\":\"image-generation\",\"prompt\":\"A simple blue circle on a white background.\",\"size\":\"1024x1024\"},\"pricing_notes\":[\"Standard token rates: $5 text input, $1.25 cached text input, $8 image input, $2 cached image input, and $30 image output per 1M tokens.\",\"Generation can run past 130s. api.chinaapi.ai reaches the origin directly with no edge timeout, so the customer path is unaffected by the Cloudflare cutoff seen on the dash host.\"]}","login_required":0,"icon":"OpenAI","tags":"OpenAI,image","vendor_id":19,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.25,"image_ratio":1.6,"enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":1250000,"image_input":8000000,"image_output":30000000,"input":5000000,"output":30000000},"paid":{"cache_read":1000000,"image_input":6400000,"image_output":24000000,"input":4000000,"output":24000000},"route_count":2}},{"model_name":"gpt-image-2.5-flare","description":"OpenAI GPT Image 2.5 Flare — an image generation and editing model from OpenAI's GPT Image 2.5 line. Flare and gpt-image-2.5-sunburst are separate builds, not aliases of one model, sold at the same price; public leaderboards score them separately. Standard token pricing is $5 text input, $1.25 cached input, $8 image input, and $30 image output per 1M tokens.","specs":"{\"type\": \"image\", \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"image\"], \"resolutions\": [\"1024x1024\", \"1536x1024\", \"1024x1536\", \"auto\"], \"health_probe\": {\"endpoint_type\": \"image-generation\", \"prompt\": \"A simple blue circle on a white background.\", \"size\": \"1024x1024\"}, \"pricing_notes\": [\"Standard: $5 text input, $1.25 cached input, $8 image input, and $30 image output per 1M tokens.\"]}","login_required":0,"tags":"OpenAI,image","vendor_id":19,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.25,"image_ratio":1.6,"enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":5,"list_input_usd":5,"selling_usd":5,"unit":"text","list_source":"https://developers.openai.com/api/docs/pricing","verified_at":"2026-09-22","faces":[{"face":"input","list_usd":5,"selling_usd":5,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":30,"selling_usd":30,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":1.25,"selling_usd":1.25,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["image_input","image_output"]},"book_faces":{"standard":{"cache_read":1250000,"image_input":8000000,"image_output":30000000,"input":5000000,"output":30000000},"paid":{"cache_read":1250000,"image_input":6400000,"image_output":24000000,"input":4000000,"output":24000000},"route_count":2}},{"model_name":"qwen-image-3.0-pro","description":"Alibaba Qwen-Image-3.0-Pro — the professional tier of the Qwen image line, built for dense, information-heavy layouts (newspapers, storyboards, menus, exam papers) in a single generation. Accepts prompts up to 4,500 tokens, renders 10px type, 12 languages, and 20+ typefaces, returns up to 6 images per request, and generates up to 2048x2048. Text-to-image and image editing.","login_required":0,"tags":"Image,Editing,High-definition","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.04,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.04,"list_input_usd":0.04,"selling_usd":0.04,"unit":"per_call","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.04,"selling_usd":0.04,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":40000},"paid":null,"route_count":1}},{"model_name":"qwen-mt-image-2.0","description":"Alibaba Qwen-MT-Image 2.0 — image translation: rewrites the text inside an image into another language while preserving layout, fonts, and the surrounding artwork, across 55 languages in any direction. Call it on the OpenAI images/edits endpoint with a public http(s) image URL in the image field plus target_lang (required; ISO code or English name) and source_lang (optional, default auto); prompt is required by the endpoint shape but ignored. Optional ext object passes the vendor's domainHint, sensitives, terminologies, and config.imageSegment through unchanged. Returns one same-size JPG URL valid for 24 hours. Input 15-8192 px per side, aspect ratio 1:10 to 10:1, up to 100 MB; uploads and data URLs are not accepted. The vendor rate-limits this model to 1 request per minute and bills an image even when it finds no translatable text (the original is returned). Served on a single official Alibaba channel.","specs":"{\"type\": \"image\", \"input_modalities\": [\"image\"], \"output_modalities\": [\"image\"], \"pricing_notes\": [\"$0.0006 per output image; a call whose image has no translatable text is still billed by the vendor and by us.\", \"Vendor rate limit: 1 request per minute.\"]}","login_required":0,"tags":"Image,Editing,Multilingual","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.0006,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.0006,"list_input_usd":0.0006,"selling_usd":0.0006,"unit":"per_call","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-07","faces":[{"face":"input","list_usd":0.0006,"selling_usd":0.0006,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":600},"paid":null,"route_count":1}},{"model_name":"qwen3.6-plus","description":"Alibaba Qwen3.6-Plus — balanced model for general reasoning and tool use, with a 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 65,536 output tokens, and an 81,920-token chain-of-thought budget. Accepts text, image, and video input and returns text. Supports function calling, structured outputs, web search, prefix completion, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.25,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.2,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.5,"list_input_usd":0.5,"selling_usd":0.5,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.5,"selling_usd":0.5,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":3,"selling_usd":3,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.625,"selling_usd":0.625,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":100000,"cache_write":625000,"input":500000,"output":3000000},"paid":{"cache_read":85000,"cache_write":531250,"input":425000,"output":2550000},"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":400000,"cache_write":2500000,"input":2000000,"output":6000000},"paid":{"cache_read":340000,"cache_write":2125000,"input":1700000,"output":5100000},"when":[{"dim":"context_len","context_len_gt":256000}]},{"tier":"standard","standard":{"cache_read":100000,"cache_write":625000,"input":500000,"output":3000000},"paid":{"cache_read":85000,"cache_write":531250,"input":425000,"output":2550000}}]}},{"model_name":"speech-2.8-turbo","description":"MiniMax speech-2.8-turbo — the low-latency tier of the 2.8 speech line, trading the HD model's ultra-realistic rendering for faster synthesis with natural flow. Covers the same 40 languages and 7 emotions, with configurable pitch, speech rate, volume, sample rate, bitrate, and output format, and accepts up to 1M characters asynchronously or 10,000 characters over the streaming interface.","login_required":0,"tags":"Audio,TTS,Multilingual,Low-latency","vendor_id":13,"quota_type":0,"model_ratio":30,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"alloy","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":60000000,"output":0},"paid":null,"route_count":1}},{"model_name":"stepaudio-2-asr-pro","description":"StepFun StepAudio 2 ASR Pro — 32B-parameter speech recognition model, the accuracy-first option in the StepFun ASR line where stepaudio-2.5-asr is the speed-and-cost option. Accepts ogg, mp3, and wav uploads; transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":2.4691335,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","book_faces":{"standard":{"input":4938267,"output":0},"paid":null,"route_count":1}},{"model_name":"gemini-3.8-flash","description":"Google Gemini 3.8 Flash — Google's fast multimodal Gemini model, callable here through both the OpenAI-compatible and the native Gemini endpoints. Standard token pricing is $0.75 input, $0.075 cached input, and $3.75 output per 1M tokens.","specs":"{\"type\": \"llm\", \"context_window\": 1048576, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\", \"video\", \"audio\", \"pdf\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"minimal\", \"low\", \"medium\", \"high\"], \"pricing_notes\": [\"Standard: $0.75 input, $0.075 cached input, and $3.75 output per 1M tokens.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Files","vendor_id":21,"quota_type":0,"model_ratio":0.375,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["openai","gemini"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.75,"list_input_usd":0.75,"selling_usd":0.75,"unit":"text","list_source":"https://linkai.llc/pricing","verified_at":"2026-09-12","faces":[{"face":"input","list_usd":0.75,"selling_usd":0.75,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":3.75,"selling_usd":3.75,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.075,"selling_usd":0.07500000000000001,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":75000,"input":750000,"output":3750000},"paid":{"cache_read":52500,"input":525000,"output":2625000},"route_count":2}},{"model_name":"glm-tts","description":"Zhipu GLM-TTS — text-to-speech built on a two-stage generation architecture with GRPO reinforcement learning, which infers emotion and intonation from the surrounding text instead of requiring explicit markup. Ships seven built-in voices (tongtong by default, xiaochen, chuichui, jam, kazi, douji, luodo) with adjustable speed and volume, accepts up to 1024 characters per request, and streams with first-frame latency under 400 ms. Returns wav or pcm at a recommended 24 kHz sample rate; streaming is pcm only.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice","vendor_id":2,"quota_type":0,"model_ratio":17.5,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"tongtong","default_response_format":"wav","speech_input":"voice-name","book_faces":{"standard":{"input":35000000,"output":0},"paid":{"input":31500000,"output":0},"route_count":1}},{"model_name":"step-tts-2","description":"StepFun step-tts-2 — end-to-end NTP-based text-to-speech built for accurate accent reproduction. Speaks Chinese, English, mixed Chinese-English, and Japanese, plus Cantonese and Sichuanese. Emotion and delivery are directed through natural-language labels, and zero-shot voice cloning takes roughly 10 seconds of reference audio, with the emotion and style controls carrying over to the cloned voice at no extra cost. Outputs wav, mp3, flac, opus, or pcm.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice,VoiceClone","vendor_id":18,"quota_type":0,"model_ratio":20.74075,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"cixingnansheng","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":41481500,"output":0},"paid":null,"route_count":1}},{"model_name":"stepaudio-3-music-preview","description":"StepFun StepAudio 3 Music (preview) — text-to-music from the StepAudio 3 series on POST /v1/music/generations, with the same request shape as minimax-music-v3.0: describe the style in prompt, add lyrics or set is_instrumental, and get the finished song back hex-encoded in data.audio (mp3 by default; wav, flac, opus or pcm through audio_setting.format). The call is synchronous and a song usually takes one to several minutes, so give the client a timeout of at least 600 seconds. Covers and accompaniment for an uploaded vocal are not offered. Free while StepFun's preview runs; StepFun retires the preview name when the free period ends and releases a paid version, so plan for this model ID to stop working at that point.","specs":"{\"type\": \"music\", \"input_modalities\": [\"text\"], \"output_modalities\": [\"audio\"], \"pricing_notes\": [\"Free while StepFun's preview runs; per call.\"]}","login_required":0,"tags":"Audio,Music,Text-to-Music","vendor_id":18,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["music-generation"],"capabilities":["audio-speech"],"tasks":["music-generation"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":0},"paid":null,"route_count":1}},{"model_name":"vidu-video-q3-turbo","description":"Vidu Q3 Turbo — the fast, cost-efficient Q3 tier for text, image, start/end-frame, and reference video generation with synchronized audio/video output. Generates 1–16 seconds for normal modes or 3–16 seconds for reference mode at 540P–1080P.","specs":"{\"type\": \"video\", \"min_duration_seconds\": 1, \"max_duration_seconds\": 16, \"resolutions\": [\"540P\", \"720P\", \"1080P\"], \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"video\", \"audio\"]}","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,First-Last-Frame,Reference-to-Video,Audio,Fast","vendor_id":26,"quota_type":1,"model_ratio":0,"model_price":0.055,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.055,"list_input_usd":0.055,"selling_usd":0.055,"unit":"per_second","list_source":"https://platform.vidu.com/docs/pricing","verified_at":"2026-09-12","faces":[{"face":"input","list_usd":0.055,"selling_usd":0.055,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"video_second":55000},"paid":null,"route_count":1}},{"model_name":"glm-image","description":"Zhipu GLM-Image — image generation built on a hybrid autoregressive-understanding plus diffusion-decoding architecture, the first SOTA multimodal model trained end-to-end on Chinese-made chips (Huawei Ascend). Reads instructions rather than keywords and fills in detail the prompt leaves implicit, with markedly stronger text rendering (Chinese characters in particular) and knowledge-intensive scenes. One image per request.","login_required":0,"tags":"Image,Text Rendering","vendor_id":2,"quota_type":1,"model_ratio":0,"model_price":0.015,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.015,"list_input_usd":0.015,"selling_usd":0.015,"unit":"per_call","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.015,"selling_usd":0.015,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":15000},"paid":{"per_call":14250},"route_count":1}},{"model_name":"qwen3.6-27b","description":"Alibaba Qwen3.6-27B — the 27-billion-parameter native vision-language dense model of the Qwen3.6 line, open-weight, which the vendor positions above 3.5-27B on agentic coding, STEM and reasoning, and on spatial intelligence, object localisation and detection. 262,144-token context window. Accepts text, image, and video input and returns text.","specs":"{\"type\":\"llm\",\"context_window\":262144,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Open Weights,256K","vendor_id":4,"quota_type":0,"model_ratio":0.3,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.6,"list_input_usd":0.6,"selling_usd":0.6,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.6,"selling_usd":0.6,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":3.6,"selling_usd":3.5999999999999996,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read"]},"book_faces":{"standard":{"input":600000,"output":3600000},"paid":null,"route_count":1}},{"model_name":"step-3.7-flash","description":"StepFun step-3.7-flash — sparse MoE with 198B total and 11B active parameters, native 256K context, and native understanding of images and video alongside text. Reasoning depth is selectable per request through reasoning_effort (low / medium / high), and the model supports reliable multi-step tool calling, JSON Schema structured output, streaming, and prompt caching. Tuned for agent and coding workloads. On /v1/responses, send the whole conversation in the input array on every turn: no conversation state is kept upstream, so requests that carry previous_response_id or conversation are refused.","login_required":0,"tags":"Vision,Video,Reasoning,Tools,256K","vendor_id":18,"quota_type":0,"model_ratio":0.1,"model_price":0,"owner_by":"","completion_ratio":5.75,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.2,"list_input_usd":0.2,"selling_usd":0.2,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.2,"selling_usd":0.2,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":1.15,"selling_usd":1.1500000000000001,"discount_pct":-0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.04,"selling_usd":0.04000000000000001,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":40000,"input":200000,"output":1150000},"paid":null,"route_count":1}},{"model_name":"stepaudio-2.5-asr","description":"StepFun StepAudio 2.5 ASR — 4B-parameter speech recognition model built on Multi-Token Prediction, transcribing five minutes of audio in about one second (RTF around 0.0053). Optimized for Chinese and English. Includes inverse text normalization, speaker identification, and dual-channel separation for call and meeting recordings. Accepts ogg, mp3, and wav uploads; transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":0.1833335,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","book_faces":{"standard":{"input":366667,"output":0},"paid":null,"route_count":1}},{"model_name":"stepaudio-3-gen-preview","description":"StepFun StepAudio 3 Audio Generation (preview) — scripted audio from the StepAudio 3 series on POST /v1/audio/generate: define roles with a name and a voice description in plain language, write the script line by line (lines in square brackets become sound effects or background music), add an overall instruction, and get the finished audio back as bytes (mp3 by default). Voices come from the role descriptions; preset voice names are not accepted. Request fields follow StepFun's API reference. Free while StepFun's preview runs; StepFun retires the preview name when the free period ends and releases a paid version, so plan for this model ID to stop working at that point.","specs":"{\"type\": \"audio-generation\", \"input_modalities\": [\"text\"], \"output_modalities\": [\"audio\"], \"pricing_notes\": [\"Free while StepFun's preview runs; per call.\"]}","login_required":0,"tags":"Audio,Sound,Voice","vendor_id":18,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-generation"],"capabilities":["audio-speech"],"tasks":["audio-generation"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":0},"paid":null,"route_count":1}},{"model_name":"claude-sonnet-4-5","description":"Anthropic Claude Sonnet 4.5 — the Sonnet 4.5 release, which Anthropic now lists among its legacy models and keeps available, for workloads that want to stay on this model version. It supports text and image input, extended thinking, a 200,000-token context window, and up to 64,000 output tokens.","specs":"{\"type\": \"llm\", \"context_window\": 200000, \"max_output_tokens\": 64000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\"], \"pricing_notes\": [\"Standard: $3 input, $0.30 cache hit, $3.75 5-minute cache write, and $15 output per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal","vendor_id":20,"quota_type":0,"model_ratio":1.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":300000,"cache_write":3750000,"cache_write_1h":6000000,"input":3000000,"output":15000000},"paid":{"cache_read":210000,"cache_write":2625000,"cache_write_1h":4200000,"input":2100000,"output":10500000},"route_count":1}},{"model_name":"doubao-seedance-2-0-260128","description":"ByteDance Seedance 2.0 — the flagship of the Seedance 2.0 line and its only member that reaches 1080p and 4K (10-bit depth) on top of 480p and 720p, at 24 fps and 4-15 seconds. Covers text-to-video, first-frame and first-and-last-frame image-to-video, multimodal reference generation from 1-9 images and up to 3 reference videos totalling 15 seconds, video editing, video extension, sound-on generation, and a web-search tool; an audio reference has to accompany an image or video. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16, with the closing frame available as an image. Billed by output resolution.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,4K","vendor_id":5,"quota_type":0,"model_ratio":3.5,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":7000000,"output":0},"paid":null,"route_count":1}},{"model_name":"gemini-3.7-flash","description":"Google Gemini 3.7 Flash — the newest Flash-tier multimodal reasoning model, priced identically to 3.6 Flash while the vendor promotion runs. Accepts text, images, video, audio and PDFs; supports thinking, function calling, code execution, search grounding and structured output.\n\nPricing note: Google's list price for this model is promotional through December 31, 2026 — $0.75 input / $3.75 output / $0.075 cached input per 1M tokens. From January 1, 2027 the vendor's list price returns to $1.50 / $7.50 / $0.15, and our price follows it at the same parity.","specs":"{\"type\": \"llm\", \"context_window\": 1048576, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\", \"video\", \"audio\", \"pdf\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"medium\", \"high\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Files","vendor_id":21,"quota_type":0,"model_ratio":0.375,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"audio_input":750000,"cache_read":75000,"input":750000,"output":3750000},"paid":{"audio_input":750000,"cache_read":52500,"input":525000,"output":2625000},"route_count":1}},{"model_name":"jev-1.13","description":"TypeSafe Jev 1.13 — a System One decision model, not a chat model. Send your application's state (a string, JSON object or array) with a set of typed questions on POST /v1/systemone — choice (pick one of your options), score (place it on your ordered levels) or noul (the probability that a yes/no statement holds) — and get typed answers back with a probability for every option and a confidence value, usually within 70–500 ms. All questions are evaluated in parallel against the same state, so several can share one request. The official TypeSafe Python and JavaScript SDKs work unchanged with the base URL set to https://api.chinaapi.ai. Billed per input token; output is free. Text input only, up to 32K tokens per request; English is the primary language, so test other languages before relying on them.","specs":"{\"type\": \"decision\", \"context_window\": 32000, \"input_modalities\": [\"text\"], \"output_modalities\": [\"decisions\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/typesafe.png","tags":"Text,Low-latency,Low-cost","vendor_id":27,"quota_type":0,"model_ratio":0.0231,"model_price":0,"owner_by":"","completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["system-one"],"capabilities":["decision"],"tasks":["structured-decision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":46200,"output":0},"paid":null,"route_count":1}},{"model_name":"qwen3.5-ocr","description":"Alibaba Qwen3.5-OCR — the OCR member of the Qwen3.5 line, built for document parsing, text localisation, and key-information extraction, with the vendor citing marked gains on real-world identity and licence documents. Accepts text and image input and returns text.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Vision,Files","vendor_id":4,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":3.5,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":100000,"output":350000},"paid":null,"route_count":1}},{"model_name":"agnes-2.5-flash","description":"Agnes AI Agnes 2.5 Flash — a fast multimodal agent model with a 512K input context and 65.5K reference output limit, supporting chat, streaming, tool calling, coding, reasoning, multi-turn dialogue, image understanding, and agent workflows.","specs":"{\"type\":\"llm\",\"context_window\":524288,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Reasoning,Tools,Vision,Multimodal,Fast","vendor_id":25,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":0,"cache_write":0,"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"claude-fable-5-1","description":"Anthropic Claude Fable 5.1 — Anthropic’s latest generally available flagship for ambitious, long-running knowledge work and coding. It is built for multi-day agentic tasks, complex software engineering, deep research, document and vision reasoning, and proactive self-verification. Cache reads are priced at 2.5% of input — the lowest of its generation.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"low\", \"medium\", \"high\", \"max\"], \"pricing_notes\": [\"Standard: $10 input, $50 output, $0.25 cache read, $12.50 5-minute cache write, and $20 1-hour cache write per 1M tokens.\", \"Cache reads are 0.025x the input price on this model, not the usual 0.1x.\", \"Adaptive thinking is always enabled.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":20,"quota_type":0,"model_ratio":5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.025,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":10,"list_input_usd":10,"selling_usd":10,"unit":"text","list_source":"https://platform.claude.com/docs/en/about-claude/pricing","verified_at":"2026-09-12","faces":[{"face":"input","list_usd":10,"selling_usd":10,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":50,"selling_usd":50,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.25,"selling_usd":0.25,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":12.5,"selling_usd":12.5,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_write_1h"]},"book_faces":{"standard":{"cache_read":250000,"cache_write":12500000,"cache_write_1h":20000000,"input":10000000,"output":50000000},"paid":{"cache_read":200000,"cache_write":10000000,"cache_write_1h":16000000,"input":8000000,"output":40000000},"route_count":2}},{"model_name":"glm-5v-turbo","description":"Zhipu GLM-5V-Turbo — multimodal reasoning model for images, video, files, coding, and tool-driven agent workflows.","login_required":0,"tags":"Reasoning,Tools,Vision,Video,Files,200K","vendor_id":2,"quota_type":0,"model_ratio":0.6,"model_price":0,"owner_by":"","completion_ratio":3.3333333333333335,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":240000,"input":1200000,"output":4000000},"paid":{"cache_read":240000,"input":1140000,"output":3800000},"route_count":1}},{"model_name":"mimo-v2.6-pro","description":"Xiaomi MiMo-V2.6-Pro — Xiaomi's trillion-parameter flagship reasoning model with open weights, natively full-modality: one model understands images, video, audio and text. 1M context and 128K max output, with deep thinking, function calling and structured output. Built for complex projects, long-horizon agent tasks, cybersecurity and research work.","login_required":0,"tags":"Reasoning,Tools,Files,Vision,Video,Audio,Multimodal,Open Weights,1M","vendor_id":16,"quota_type":0,"model_ratio":0.2175,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.008275862068965517,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic","openai-response"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","reasoning","vision","file-video-understanding","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.435,"list_input_usd":0.435,"selling_usd":0.435,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-09-22","faces":[{"face":"input","list_usd":0.435,"selling_usd":0.435,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.87,"selling_usd":0.87,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.0036,"selling_usd":0.0036,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":3600,"input":435000,"output":870000},"paid":null,"route_count":1}},{"model_name":"doubao-seedream-5-0-pro-260628","description":"ByteDance Doubao Seedream 5.0 Pro — the vendor's top image model for controllable creation: text-to-image, single-reference and multi-reference image-to-image (2-10 reference images), interactive editing by coordinates, boxes, or arrows, and layer decomposition that splits one image into a base plus up to 16 layers, each returned with its own size, z_index, and bounding box (set layer_decomposition=true). Single-image output only: no group (sequential) generation, web search, or streaming. Billed per output image by scene and pixel tier: single image at or under 2.61 MP is the base price; above 2.61 MP costs 2x; in layer decomposition every returned layer is billed separately at 0.5x (or 1x above 2.61 MP); reference images beyond the first add 1/15 of the base price each.","specs":"{\"type\": \"image\", \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"image\"], \"max_reference_images\": 10, \"max_layers\": 16, \"pricing_notes\": [\"Base price: single image generation at or under 2.61 megapixels (1.5K or lower).\", \"Single image above 2.61 MP: 2x base. Layer decomposition: 0.5x base per returned layer (1x above 2.61 MP).\", \"Reference images: first free, each additional image adds 1/15 of the base price.\"]}","login_required":0,"tags":"Image,Editing,2K","vendor_id":5,"quota_type":1,"model_ratio":0,"model_price":0.045,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.045,"list_input_usd":0.045,"selling_usd":0.045,"unit":"per_call","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-09-06","faces":[{"face":"input","list_usd":0.045,"selling_usd":0.045,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"per_call":45000},"paid":null,"route_count":1}},{"model_name":"gemini-2.5-flash-image","description":"Google Gemini 2.5 Flash Image (Nano Banana) — a fast native image generation and editing model for creative workflows. It takes text and image input and returns images through the OpenAI-compatible chat completions endpoint, with a 65,536-token input limit and 32,768-token output limit.","specs":"{\"type\": \"image\", \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"image\"], \"resolutions\": [\"1024x1024\"], \"request_protocol\": \"OpenAI Chat Completions\", \"image_response_path\": \"choices[0].message.content\", \"health_probe\": {\"endpoint_type\": \"openai\", \"prompt\": \"A simple blue circle on a white background.\"}, \"pricing_notes\": [\"Standard: $0.30 text/image input and $2.50 text output per 1M tokens; image output is carried as img_o * 30 per 1M image tokens.\", \"65,536 input and 32,768 output token limits, smaller than the gemini-2.5-flash sibling.\"]}","login_required":0,"icon":"Gemini.Color","tags":"Google,image","vendor_id":21,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":8.333333333333334,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","gemini"],"capabilities":["text-multimodal","image"],"tasks":["chat","text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"image_output":30000000,"input":300000,"output":2500000},"paid":{"image_output":21000000,"input":210000,"output":1750000},"route_count":1}},{"model_name":"gpt-6-astra","description":"OpenAI GPT-6 Astra — OpenAI's most capable model, built for the hardest end-to-end work: complex reasoning, coding, computer use, research, and document creation. It supports text and image input, a 1,050,000-token context window, up to 128,000 output tokens, and reasoning effort settings up to xhigh and max.","specs":"{\"type\": \"llm\", \"context_window\": 1050000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"low\", \"medium\", \"high\", \"xhigh\", \"max\"], \"pricing_notes\": [\"Standard: $10 input, $1 cached input, $12.50 cache write, and $50 output per 1M tokens.\", \"Above 272K input tokens, the full request uses 2x input and cache rates and 1.5x output pricing.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":19,"quota_type":0,"model_ratio":5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":10,"list_input_usd":10,"selling_usd":10,"unit":"text","list_source":"https://developers.openai.com/api/docs/pricing","verified_at":"2026-09-05","faces":[{"face":"input","list_usd":10,"selling_usd":10,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":50,"selling_usd":50,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":1,"selling_usd":1,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":12.5,"selling_usd":12.5,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":1000000,"cache_write":12500000,"input":10000000,"output":50000000},"paid":{"cache_read":1000000,"cache_write":8750000,"input":7000000,"output":35000000},"route_count":3,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":2000000,"cache_write":25000000,"input":20000000,"output":75000000},"paid":{"cache_read":2000000,"cache_write":17500000,"input":14000000,"output":52500000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":1000000,"cache_write":12500000,"input":10000000,"output":50000000},"paid":{"cache_read":1000000,"cache_write":8750000,"input":7000000,"output":35000000}}]}},{"model_name":"qwen3.7-flash","description":"Alibaba Qwen3.7-Flash — the fast tier of the Qwen3.7 native vision-language line, which the vendor highlights for multimodal agent work (search and CI agents) and for steadier end-to-end task execution than 3.6-Flash. 1,000,000-token context window, up to 65,536 output tokens, and up to 256 images or 64 videos per request. Accepts text, image, and video input and returns text; supports thinking mode, function calling, built-in tools, and structured outputs. Priced in three context bands, so a long-context request costs more per token than a short one.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Vision,1M,Fast","vendor_id":4,"quota_type":0,"model_ratio":0.015,"model_price":0,"owner_by":"","completion_ratio":4.333333333333333,"cache_ratio":0.2,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.03,"list_input_usd":0.03,"selling_usd":0.03,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.03,"selling_usd":0.03,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.13,"selling_usd":0.12999999999999998,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.006,"selling_usd":0.006,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":0.0375,"selling_usd":0.0375,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":6000,"cache_write":37500,"input":30000,"output":130000},"paid":null,"route_count":1,"tier_count":3,"tiers":[{"tier":"context_256k_1m","standard":{"cache_read":40000,"cache_write":250000,"input":200000,"output":800000},"paid":null,"when":[{"dim":"context_len","context_len_gt":256000}]},{"tier":"context_32k_256k","standard":{"cache_read":20000,"cache_write":125000,"input":100000,"output":400000},"paid":null,"when":[{"dim":"context_len","context_len_gt":32000}]},{"tier":"standard","standard":{"cache_read":6000,"cache_write":37500,"input":30000,"output":130000},"paid":null}]}},{"model_name":"stepaudio-2.5-chat","description":"StepFun StepAudio 2.5 Chat — end-to-end speech understanding model: it takes audio or text in and returns text, reading paralinguistic signal such as hesitation and laughter from the waveform itself rather than from a transcript. Supports extensive persona customization covering character, speech habits, and emotional range. Audio is supplied as input_audio content parts on the chat completions endpoint. For spoken replies use a TTS model.","login_required":0,"tags":"Audio,Reasoning,Tools","vendor_id":18,"quota_type":0,"model_ratio":0.75,"model_price":0,"owner_by":"","completion_ratio":2.3333333333333335,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","reasoning","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.5,"list_input_usd":1.5,"selling_usd":1.5,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":1.5,"selling_usd":1.5,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":3.5,"selling_usd":3.5,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.3,"selling_usd":0.30000000000000004,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":300000,"input":1500000,"output":3500000},"paid":null,"route_count":1}},{"model_name":"gemini-3.5-flash","description":"Google Gemini 3.5 Flash — a stable multimodal model optimized for sustained agentic performance, rapid coding loops, sub-agent deployment, and long-running multi-step workflows. It accepts text, images, video, audio, and PDFs, supports thinking and built-in tools, and provides a 1,048,576-token input context with up to 65,536 output tokens.","specs":"{\"type\":\"llm\",\"context_window\":1048576,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\",\"audio\",\"pdf\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"minimal\",\"low\",\"medium\",\"high\"],\"pricing_notes\":[\"Standard: $1.50 input, $3.00 audio input, $0.15 cached input, and $9.00 output per 1M tokens.\"]}","login_required":0,"icon":"Gemini.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Files","vendor_id":21,"quota_type":0,"model_ratio":0.75,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai","gemini","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"audio_input":1500000,"cache_read":150000,"input":1500000,"output":9000000},"paid":{"audio_input":1500000,"cache_read":105000,"input":1050000,"output":6300000},"route_count":3}},{"model_name":"kimi-k3","description":"Moonshot Kimi K3 — 2.8-trillion-parameter flagship for long-horizon coding, end-to-end knowledge work, and deep reasoning, with a 1M-token context window and up to 1,048,576 completion tokens (131,072 by default). Accepts text, base64-encoded images, and base64-encoded video and returns text. Thinking is always on, with reasoning effort at max by default; temperature is fixed at 1.0. Supports custom tool calling and dynamic tool loading.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 131072, \"input_modalities\": [\"text\", \"image\", \"video\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"low\", \"high\", \"max\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,1M","vendor_id":3,"quota_type":0,"model_ratio":1.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":3,"list_input_usd":3,"selling_usd":3,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":3,"selling_usd":3,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":15,"selling_usd":15,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.3,"selling_usd":0.30000000000000004,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":300000,"input":3000000,"output":15000000},"paid":{"cache_read":300000,"input":3000000,"output":15000000},"route_count":3}},{"model_name":"mimo-v2.5-tts-voiceclone","description":"Xiaomi MiMo v2.5 TTS VoiceClone — voice cloning model: supply a reference audio sample as a data URL in the voice field (data:{mime};base64,...) and it synthesises speech in that voice with no training, annotation, or fine-tuning step. Accepts MP3 (audio/mpeg) or WAV, with the base64-encoded string capped at 10 MB; Xiaomi publishes no minimum reference duration. Natural-language style instructions and inline audio tags carry over from mimo-v2.5-tts, but streaming is not available — the request runs in compatibility mode and returns once synthesis has completed.","login_required":0,"tags":"Audio,TTS,VoiceClone","vendor_id":16,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_response_format":"wav","speech_input":"voice-clone","book_faces":{"standard":{"input":0,"output":0},"paid":null,"route_count":1}},{"model_name":"qwen-audio-3.0-realtime-flash","description":"Alibaba Qwen-Audio-3.0-Realtime-Flash — full-duplex speech-to-speech conversation over a WebSocket session, from the line the vendor reports as first overall in the Artificial Analysis Speech-to-Speech ranking. The flash tier is tuned for the lowest end-to-end response latency. Reached over GET /v1/realtime. Audio input and audio output are billed on their own faces, well above the text rate.","specs":"{\"type\":\"llm\",\"input_modalities\":[\"text\",\"audio\"],\"output_modalities\":[\"text\",\"audio\"]}","login_required":0,"tags":"Realtime,Audio,Voice,Multilingual,Fast","vendor_id":4,"quota_type":0,"model_ratio":0.225,"model_price":0,"owner_by":"","completion_ratio":10,"cache_ratio":1,"audio_ratio":10,"audio_completion_ratio":33.333333333333336,"enable_groups":["default"],"supported_endpoint_types":["realtime"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.45,"list_input_usd":0.45,"selling_usd":0.45,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-04","faces":[{"face":"input","list_usd":0.45,"selling_usd":0.45,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":4.5,"selling_usd":4.5,"discount_pct":0,"verdict":"at_list"}],"unverified_faces":["cache_read","audio_input","audio_output"]},"book_faces":{"standard":{"audio_input":4500000,"audio_output":15000000,"input":450000,"output":4500000},"paid":null,"route_count":1}},{"model_name":"stepaudio-3-asr-max","description":"StepFun StepAudio 3 ASR Max — StepFun's largest speech-recognition model, built on a large language model so recognition draws on context and domain knowledge: stronger on names, drug names, technical terms and homophones, on mixed Chinese-English, dialects and long audio, and on whispered, very fast or music-backed speech, including sung lyrics. Send an audio file to POST /v1/audio/transcriptions (WAV, MP3 or OGG); billed by audio duration.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":3.3333335,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"USD","list_input":0.4,"list_input_usd":0.4,"selling_usd":0.40000002,"unit":"asr","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-09-21","faces":[{"face":"input","list_usd":0.4,"selling_usd":0.40000002,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"input":6666667,"output":0},"paid":null,"route_count":1}},{"model_name":"agnes-2.5-pro-beta","description":"Agnes AI Agnes 2.5 Pro Beta — the low-priced tier of the Pro family, billed at its own Beta price well below 2.5 Pro's, a paid reasoning model with the same request format, text protocols and context limits, for advanced coding, scientific reasoning, long-context analysis, agent workflows and image understanding.","specs":"{\"type\": \"llm\", \"context_window\": 1048576, \"max_output_tokens\": 65536, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Reasoning,Tools,Vision,Multimodal,1M","vendor_id":25,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.1,"create_cache_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"text","list_source":"https://www.agnes-ai.com/zh-Hans/docs/agnes-25-pro-beta","verified_at":"2026-09-02","faces":[{"face":"input","list_usd":0.1,"selling_usd":0.1,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":0.3,"selling_usd":0.30000000000000004,"discount_pct":-0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.01,"selling_usd":0.010000000000000002,"discount_pct":-0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":10000,"cache_write":0,"input":100000,"output":300000},"paid":null,"route_count":1}},{"model_name":"agnes-video-2.5-flash","description":"Agnes AI Video 2.5 Flash — the 720P-only variant of Video 2.5, generating 4-12 second video from text, first/last keyframes, or image and audio references, with the same asynchronous task API.","specs":"{\"type\":\"video\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"video\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Video,Text-to-Video,Image-to-Video","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","book_faces":{"standard":{"video_second":0},"paid":null,"route_count":1}},{"model_name":"claude-opus-4-6","description":"Anthropic Claude Opus 4.6 — the Opus generation for complex reasoning and agentic work, with a 1,000,000-token context window at standard pricing and up to 128,000 output tokens. It supports text and image input and adaptive thinking, and uses the pre-4.7 tokenizer.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\"], \"pricing_notes\": [\"Standard: $5 input, $0.50 cache hit, $6.25 5-minute cache write, and $25 output per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M","vendor_id":20,"quota_type":0,"model_ratio":2.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["anthropic","openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":500000,"cache_write":6250000,"cache_write_1h":10000000,"input":5000000,"output":25000000},"paid":{"cache_read":500000,"cache_write":4375000,"cache_write_1h":7000000,"input":3500000,"output":17500000},"route_count":5},"route_health":{"window_hours":24,"sample_size":13,"success_ratio":1,"last_checked_at":1790603554}},{"model_name":"kimi-k2.6","description":"Moonshot Kimi K2.6 — general-purpose model for dialogue, code generation, visual understanding, and agent tasks, with stronger long-horizon code writing and autonomous execution than its predecessor. 256K context window, 32,768 default output tokens. Accepts text, images (png, jpeg, webp, gif), and video (mp4, mov, webm and more) as base64 content and returns text. Thinking mode is on by default and can be disabled; temperature is fixed at 1.0 with thinking and 0.6 without.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,256K","vendor_id":3,"quota_type":0,"model_ratio":0.475,"model_price":0,"owner_by":"","completion_ratio":4.2105263157894735,"cache_ratio":0.16842105263157894,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.95,"list_input_usd":0.95,"selling_usd":0.95,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":0.95,"selling_usd":0.95,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":4,"selling_usd":3.9999999999999996,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.16,"selling_usd":0.15999999999999998,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":160000,"input":950000,"output":4000000},"paid":{"cache_read":128000,"input":760000,"output":3200000},"route_count":2}},{"model_name":"qwen-flash-character","description":"Alibaba Qwen-Flash-Character — Qwen's multilingual role-play model (dynamic version; the vendor announces updates in advance), tuned for persona-faithful character conversation: persona-constrained instruction following, topic advancement, listening, and empathy. 8,192-token context window, text in and text out. No thinking mode, tool calling, built-in tools, or structured output. The vendor's session cache applies automatically on its side.","specs":"{\"type\": \"llm\", \"context_window\": 8192, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Multilingual,Low-cost,8K","vendor_id":4,"quota_type":0,"model_ratio":0.025,"model_price":0,"owner_by":"","completion_ratio":8,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":50000,"output":400000},"paid":null,"route_count":1}},{"model_name":"qwen3.8-max-0902","description":"Alibaba Qwen3.8-Max-0902 — the 2026-09-02 snapshot of Qwen3.8-Max (vendor alias qwen3.8-max-2026-09-02), the 2.4-trillion-parameter MoE flagship, frozen so integrations can pin this exact build; the vendor cites deeper coding for complex engineering projects and long-horizon autonomous development. 1,000,000-token context window, up to 131,072 output tokens; accepts text, image, and video input and returns text; supports thinking mode, tool calling, and structured output. The dynamic qwen3.8-max name moves to newer builds as the vendor releases them — pin this name to stay on this one. Priced identically to qwen3.8-max.","specs":"{\"type\": \"llm\", \"context_window\": 1000000, \"max_output_tokens\": 131072, \"input_modalities\": [\"text\", \"image\", \"video\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Reasoning,Tools,1M","vendor_id":4,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.125,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2,"list_input_usd":2,"selling_usd":2,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-09-13","faces":[{"face":"input","list_usd":2,"selling_usd":2,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":6,"selling_usd":6,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.25,"selling_usd":0.25,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":2.5,"selling_usd":2.5,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":250000,"cache_write":2500000,"input":2000000,"output":6000000},"paid":null,"route_count":1}},{"model_name":"speech-2.8-hd","description":"MiniMax speech-2.8-hd — the high-definition tier of the 2.8 speech line, positioned for ultra-realistic delivery with sound tags, covering 40 languages and 7 emotions. Pitch, speech rate, volume, sample rate, bitrate, and output format are configurable per request, and long-form synthesis accepts up to 1M characters asynchronously or 10,000 characters over the streaming interface.","login_required":0,"tags":"Audio,TTS,Multilingual,High-definition","vendor_id":13,"quota_type":0,"model_ratio":50,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","default_voice":"alloy","default_response_format":"mp3","speech_input":"voice-name","book_faces":{"standard":{"input":100000000,"output":0},"paid":null,"route_count":1}},{"model_name":"claude-haiku-4-5","description":"Anthropic Claude Haiku 4.5 — the fastest Claude model with near-frontier intelligence, built for high-volume, latency-sensitive work such as classification, extraction, and routing. It supports text and image input, extended thinking, a 200,000-token context window, and up to 64,000 output tokens.","specs":"{\"type\": \"llm\", \"context_window\": 200000, \"max_output_tokens\": 64000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\"], \"pricing_notes\": [\"Standard: $1 input, $0.10 cache hit, $1.25 5-minute cache write, and $5 output per 1M tokens.\"]}","login_required":0,"icon":"Claude.Color","tags":"Reasoning,Tools,Vision,Fast,Low-cost","vendor_id":20,"quota_type":0,"model_ratio":0.5,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":100000,"cache_write":1250000,"cache_write_1h":2000000,"input":1000000,"output":5000000},"paid":{"cache_read":70000,"cache_write":875000,"cache_write_1h":1400000,"input":700000,"output":3500000},"route_count":1}},{"model_name":"doubao-seedream-5-0-260128","description":"ByteDance Doubao Seedream 5.0-lite — text-to-image and image-to-image with smarter controllable creation, real-time retrieval, and stronger subject consistency (2K+ resolution).","login_required":0,"tags":"Image,4K","vendor_id":5,"quota_type":1,"model_ratio":0,"model_price":0.032593,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":32593},"paid":null,"route_count":1}},{"model_name":"gpt-6-sol","description":"OpenAI GPT-6 Sol — built for complex coding and agentic workflows. It supports text and image input, a 1,050,000-token context window, up to 128,000 output tokens, and reasoning effort from none to max, with medium as the default. Use /v1/responses for tool calling: on /v1/chat/completions, function calling works only with reasoning_effort set to none. On /v1/responses, send the whole conversation in the input array on every turn; requests that carry previous_response_id or conversation are refused.","specs":"{\"type\": \"llm\", \"context_window\": 1050000, \"max_output_tokens\": 128000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"reasoning_efforts\": [\"none\", \"low\", \"medium\", \"high\", \"xhigh\", \"max\"], \"pricing_notes\": [\"Standard: $2 input, $0.20 cached input, $2.50 cache write, and $10 output per 1M tokens.\", \"Above 272K input tokens, the full request uses 2x input and cache rates and 1.5x output pricing.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,1M,Files","vendor_id":19,"quota_type":0,"model_ratio":1,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2,"list_input_usd":2,"selling_usd":2,"unit":"text","list_source":"https://developers.openai.com/api/docs/pricing","verified_at":"2026-09-24","faces":[{"face":"input","list_usd":2,"selling_usd":2,"discount_pct":0,"verdict":"at_list"},{"face":"output","list_usd":10,"selling_usd":10,"discount_pct":0,"verdict":"at_list"},{"face":"cache_read","list_usd":0.2,"selling_usd":0.2,"discount_pct":0,"verdict":"at_list"},{"face":"cache_write","list_usd":2.5,"selling_usd":2.5,"discount_pct":0,"verdict":"at_list"}]},"book_faces":{"standard":{"cache_read":200000,"cache_write":2500000,"input":2000000,"output":10000000},"paid":{"cache_read":140000,"cache_write":1750000,"input":1400000,"output":7000000},"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":400000,"cache_write":5000000,"input":4000000,"output":15000000},"paid":{"cache_read":280000,"cache_write":3500000,"input":2800000,"output":10500000},"when":[{"dim":"context_len","context_len_gt":272000}]},{"tier":"standard","standard":{"cache_read":200000,"cache_write":2500000,"input":2000000,"output":10000000},"paid":{"cache_read":140000,"cache_write":1750000,"input":1400000,"output":7000000}}]}},{"model_name":"hy-mt2-plus","description":"Tencent HY-MT2 Plus — instruction-following multilingual translation tier through the OpenAI Chat Completions protocol. Up to 4K input tokens and 4K output tokens; text-only.","specs":"{\"type\": \"llm\", \"context_window\": 8192, \"max_input_tokens\": 4096, \"max_output_tokens\": 4096, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Translation,Multilingual,Text","vendor_id":17,"quota_type":0,"model_ratio":0.037037,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":74074,"output":296296},"paid":null,"route_count":1}},{"model_name":"doubao-seed-character-260628","description":"ByteDance Doubao Seed Character (260628) — Doubao's character-dialogue model for persona-driven, multi-turn conversation, the successor line to doubao-1-5-pro-32k-character. This build carries the vendor's deep thinking, text generation, multimodal understanding, tool calling, and structured output (json_schema recommended) capability labels. 128K context window with 96K maximum input and up to 32K output tokens (4K by default). Priced by input length: requests up to 32K input tokens bill at the standard tier; longer requests bill at the long-context tier, where input costs 1.5 times and output three times the standard-tier rate.","specs":"{\"type\": \"llm\", \"context_window\": 128000, \"max_input_tokens\": 96000, \"max_output_tokens\": 32000, \"input_modalities\": [\"text\", \"image\"], \"output_modalities\": [\"text\"], \"pricing_notes\": [\"Standard tier (input up to 32K tokens): CNY 0.80 input, 0.16 cached input, 2.00 output per 1M tokens.\", \"Long-context tier (input above 32K tokens): CNY 1.20 input, 0.16 cached input, 6.00 output per 1M tokens.\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,128K","vendor_id":5,"quota_type":0,"model_ratio":0.059259,"model_price":0,"owner_by":"","completion_ratio":2.500008437536914,"cache_ratio":0.19999493747785146,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"cache_read":23703,"input":118518,"output":296296},"paid":null,"route_count":1,"tier_count":2,"tiers":[{"tier":"long_context","standard":{"cache_read":23703,"input":177777,"output":888888},"paid":null,"when":[{"dim":"context_len","context_len_gt":32000}]},{"tier":"standard","standard":{"cache_read":23703,"input":118518,"output":296296},"paid":null}]}},{"model_name":"hy-mt2-lite","description":"Tencent HY-MT2 Lite — latency- and cost-oriented multilingual translation tier through the OpenAI Chat Completions protocol. Up to 4K input tokens and 4K output tokens; text-only.","specs":"{\"type\": \"llm\", \"context_window\": 8192, \"max_input_tokens\": 4096, \"max_output_tokens\": 4096, \"input_modalities\": [\"text\"], \"output_modalities\": [\"text\"]}","login_required":0,"tags":"Translation,Multilingual,Text,Fast","vendor_id":17,"quota_type":0,"model_ratio":0.022222,"model_price":0,"owner_by":"","completion_ratio":4.0000450004500046,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","book_faces":{"standard":{"input":44444,"output":177778},"paid":null,"route_count":1}},{"model_name":"qwen3-asr-flash","description":"Alibaba Qwen3-ASR-Flash — speech recognition built on the Qwen3 foundation model, which automatically determines the spoken language and transcribes 11 languages; the Qwen3-ASR line documents Mandarin alongside Sichuanese, Min Nan, Wu, and Cantonese, and multiple English accents. Accepts aac, amr, avi, aiff, flac, flv, mkv, mp3, mpeg, ogg, opus, wav, webm, wma, and wmv up to 5 minutes and 10 MB per request; pcm input must be 16 kHz and other formats are resampled to 16 kHz before recognition.","login_required":0,"tags":"Audio,ASR,Multilingual,Low-latency","vendor_id":4,"quota_type":0,"model_ratio":1.05,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","book_faces":{"standard":{"input":2100000,"output":0},"paid":null,"route_count":2}},{"model_name":"agnes-image-2.1-flash","description":"Agnes AI Image 2.1 Flash — a detail-focused image generation and editing model for high-information-density scenes, text-to-image, image-to-image, and multi-image composition across flexible 1K–4K sizes and aspect ratios.","specs":"{\"type\":\"image\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"image\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Image,Editing,High-definition","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","book_faces":{"standard":{"per_call":0},"paid":null,"route_count":1}}],"group_ratio":{"default":1},"paid_discount_model_count":50,"pricing_version":"5f204e0fe778f3fc61ee96a9f087237943b089e934a5d3892763ce7d2f934c5c","success":true,"supported_endpoint":{"anthropic":{"path":"/v1/messages","method":"POST"},"audio-generation":{"path":"/v1/audio/generate","method":"POST"},"audio-speech":{"path":"/v1/audio/speech","method":"POST"},"audio-transcription":{"path":"/v1/audio/transcriptions","method":"POST"},"gemini":{"path":"/v1beta/models/{model}:generateContent","method":"POST"},"image-generation":{"path":"/v1/images/generations","method":"POST"},"music-generation":{"path":"/v1/music/generations","method":"POST"},"openai":{"path":"/v1/chat/completions","method":"POST"},"openai-response":{"path":"/v1/responses","method":"POST"},"openai-video":{"path":"/v1/videos","method":"POST"},"realtime":{"path":"/v1/realtime","method":"GET"},"system-one":{"path":"/v1/systemone","method":"POST"}},"usable_group":{"default":"Default"},"vendors":[{"id":13,"name":"MiniMax","description":"MiniMax — MiniMax LLMs and Hailuo video generation.","icon":"Minimax.Color"},{"id":18,"name":"StepFun","description":"StepFun (阶跃星辰) — Step model family.","icon":"Stepfun.Color"},{"id":5,"name":"字节跳动","description":"ByteDance — Doubao LLMs and Seedance video generation.","icon":"Doubao.Color"},{"id":15,"name":"Meituan","description":"Meituan — LongCat model family.","icon":"https://chinaapi.ai/assets/logos/longcat.png"},{"id":20,"name":"Anthropic","description":"Claude models","icon":"Claude.Color"},{"id":21,"name":"Google","description":"Gemini models","icon":"Gemini.Color"},{"id":2,"name":"智谱","description":"Zhipu AI — the GLM model family.","icon":"Zhipu.Color"},{"id":16,"name":"Xiaomi","description":"Xiaomi — MiMo model family.","icon":"https://chinaapi.ai/assets/logos/xiaomi.png"},{"id":17,"name":"Tencent","description":"Tencent — Hunyuan (混元) model family.","icon":"Hunyuan.Color"},{"id":19,"name":"OpenAI","description":"OpenAI models","icon":"OpenAI"},{"id":25,"name":"Agnes AI","description":"Agnes AI — multimodal text, image, and video generation models.","icon":"https://dash.chinaapi.ai/providers/agnes.png"},{"id":26,"name":"Vidu","description":"Vidu — ShengShu Technology video generation models.","icon":"https://dash.chinaapi.ai/providers/vidu.svg"},{"id":3,"name":"Moonshot","description":"Moonshot AI — the Kimi model family.","icon":"Moonshot"},{"id":4,"name":"阿里巴巴","description":"Alibaba — the Qwen model family.","icon":"Qwen.Color"},{"id":6,"name":"快手","description":"Kuaishou — the Kling video generation family.","icon":"Kling.Color"},{"id":27,"name":"TypeSafe","description":"TypeSafe — Jev, a System One decision model: typed questions in, typed answers with calibrated probabilities out.","icon":"https://dash.chinaapi.ai/providers/typesafe.png"},{"id":1,"name":"DeepSeek","description":"DeepSeek — open-weights frontier LLMs for reasoning and coding.","icon":"DeepSeek.Color"}]}