{"auto_groups":["default"],"data":[{"model_name":"agnes-image-2.0-flash","description":"Agnes AI Image 2.0 Flash — a fast image generation and editing model for text-to-image, image-to-image, and multi-image composition, with generated results available as URLs or Base64 data.","specs":"{\"type\":\"image\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"image\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Image,Editing,Fast","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","pricing_version":"5a90f2b86c08bd983a9a2e6d66c255f4eaef9c4bc934386d2b6ae84ef0ff1f1f","billing_unit":"per_call"},{"model_name":"kling-3.0-turbo","description":"Kuaishou Kling 3.0 Turbo — the value tier of the Kling 3.0 line, generating 720P or 1080P (720P default) at 3-15 seconds (5 by default) in 16:9, 9:16, or 1:1, from a prompt or a first-frame image. Native audio is always included and has no on/off switch. Against kling-v3 it drops the 4K tier, tail-frame control, and video input in exchange for speed and cost. 720p 5s with audio ≈ $0.56.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Audio,Fast","vendor_id":6,"quota_type":1,"model_ratio":0,"model_price":0.56,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.56,"list_input_usd":0.56,"selling_usd":0.56,"unit":"per_call","list_source":"https://kling.ai/dev/pricing","verified_at":"2026-07-31"}},{"model_name":"hy3","description":"Tencent Hunyuan 3 (Hy3) — 295B-parameter (21B active) MoE model, native 256K context, three thinking modes (fast / low / deep). Strong at coding agents, long-document understanding, multi-turn context, and multi-step agent workflows. Text-only.","login_required":0,"tags":"Reasoning,Tools,256K","vendor_id":17,"quota_type":0,"model_ratio":0.076923,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.25,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"CNY","list_input":1,"list_input_usd":0.15384615384615385,"list_fx_rate":6.5,"selling_usd":0.153846,"unit":"text","list_source":"https://cloud.tencent.com/document/product/1823/130055","verified_at":"2026-08-03"}},{"model_name":"kimi-k2.6","description":"Moonshot Kimi K2.6 — general-purpose model for dialogue, code generation, visual understanding, and agent tasks, with stronger long-horizon code writing and autonomous execution than its predecessor. 256K context window, 32,768 default output tokens. Accepts text, images (png, jpeg, webp, gif), and video (mp4, mov, webm and more) as base64 content and returns text. Thinking mode is on by default and can be disabled; temperature is fixed at 1.0 with thinking and 0.6 without.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,256K","vendor_id":3,"quota_type":0,"model_ratio":0.38,"model_price":0,"owner_by":"","completion_ratio":4.153846,"cache_ratio":0.169231,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":20,"list_currency":"USD","list_input":0.95,"list_input_usd":0.95,"selling_usd":0.76,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat-k26","verified_at":"2026-08-06"}},{"model_name":"LongCat-2.0","description":"Meituan LongCat-2.0 — 1.6T-parameter open MoE model with native 1M context, built for agentic coding and long-horizon tool use. Text-only reasoning model.","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":15,"quota_type":0,"model_ratio":0.15,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.3,"unit":"text","list_source":"https://longcat.ai/platform/docs/pricing/long-cat-2.0","verified_at":"2026-08-06"}},{"model_name":"step-audio-2","description":"StepFun step-audio-2 — end-to-end speech model that consumes audio directly rather than working from a transcript, so tone and delivery survive into the model's understanding. Billed per token, sharing the input rate of the StepAudio 2.5 line at a higher output rate. StepFun publishes no separate capability page for this model; consult the platform audio documentation for the current parameter set.","login_required":0,"tags":"Audio","vendor_id":18,"quota_type":0,"model_ratio":0.775,"model_price":0,"owner_by":"","completion_ratio":7,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"above_list","discount_pct":-0.75,"list_currency":"CNY","list_input":10,"list_input_usd":1.5384615384615385,"list_fx_rate":6.5,"selling_usd":1.55,"unit":"text","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"step-tts-mini","description":"StepFun step-tts-mini — the cost-efficient, low-latency member of the StepFun TTS line, covering the same languages as step-tts-2 (Chinese, English, mixed Chinese-English, and Japanese, plus Cantonese and Sichuanese) with a subset of the built-in voices. Supports natural-language emotion and delivery labels and zero-shot voice cloning from roughly 10 seconds of reference audio. Outputs wav, mp3, flac, opus, or pcm.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice,VoiceClone,Low-latency","vendor_id":18,"quota_type":0,"model_ratio":7.5,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"above_list","discount_pct":-8.33,"list_currency":"CNY","list_input":0.9,"list_input_usd":0.13846153846153847,"list_fx_rate":6.5,"selling_usd":0.15,"unit":"tts","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"stepaudio-2.5-asr","description":"StepFun StepAudio 2.5 ASR — 4B-parameter speech recognition model built on Multi-Token Prediction, transcribing five minutes of audio in about one second (RTF around 0.0053). Optimized for Chinese and English. Includes inverse text normalization, speaker identification, and dual-channel separation for call and meeting recordings. Accepts ogg, mp3, and wav uploads; transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":0.1833333,"model_price":0,"owner_by":"","audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.022,"list_input_usd":0.022,"selling_usd":0.021999996,"unit":"asr","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"stepaudio-2.5-tts","description":"StepFun StepAudio 2.5 TTS — contextual text-to-speech that carries understanding through the whole generation pipeline instead of treating delivery as post-processing. Voice, emotion, and pacing are directed in natural language at two levels: a global context for the request and inline context for individual passages. Zero-shot voice cloning needs about 3 seconds of reference audio and inherits the full contextual control set. Accepts up to 1000 characters per request; outputs wav, mp3, flac, or opus.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice,VoiceClone","vendor_id":18,"quota_type":0,"model_ratio":42.5,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.85,"list_input_usd":0.85,"selling_usd":0.85,"unit":"tts","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"agnes-2.5-flash","description":"Agnes AI Agnes 2.5 Flash — a fast multimodal agent model with a 512K input context and 65.5K reference output limit, supporting chat, streaming, tool calling, coding, reasoning, multi-turn dialogue, image understanding, and agent workflows.","specs":"{\"type\":\"llm\",\"context_window\":524288,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Reasoning,Tools,Vision,Multimodal,Fast","vendor_id":25,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":0,"create_cache_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens"},{"model_name":"glm-5","description":"Zhipu GLM-5 — 744B-parameter MoE with 40B active, trained on 28.5T tokens and using DeepSeek Sparse Attention, with a 200K context window and up to 128K output tokens. Text in, text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":200000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"Zhipu.Color","tags":"Reasoning,Tools,Open Weights,200K","vendor_id":2,"quota_type":0,"model_ratio":0.5,"model_price":0,"owner_by":"","completion_ratio":3.2,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1,"list_input_usd":1,"selling_usd":1,"unit":"text","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-07-30"}},{"model_name":"mimo-v2.5-tts","description":"Xiaomi MiMo v2.5 TTS — expressive text-to-speech with eight built-in voices, four Chinese and four English (Mia, Chloe, Milo, Dean), covering Chinese and English plus Northeastern, Sichuanese, Henan, and Cantonese dialect styles. Delivery is directed two ways: natural-language style instructions, and inline audio tags for basic and compound emotions, breathing, sighing, and pacing, including a singing mode unique to this model in the MiMo TTS line. Returns 24 kHz mono wav or pcm16 and supports low-latency streaming, which requires pcm16. 8K context and 8K maximum output.","login_required":0,"tags":"Audio,TTS,Voice","vendor_id":16,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters"},{"model_name":"step-3.5-flash-2603","description":"StepFun step-3.5-flash-2603 — the agent-tuned 256K snapshot of step-3.5-flash, optimized for token efficiency and a low-inference operating mode. Its reasoning_effort parameter accepts only low and high, where the base model takes three levels, so code that sends medium must be adapted. Text-only; supports tool calling, JSON Schema structured output, streaming, and prompt caching.","login_required":0,"tags":"Reasoning,Tools,256K","vendor_id":18,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"step-asr","description":"StepFun step-asr — general-purpose speech recognition covering Chinese, English, and Chinese dialects, for transcription, meeting minutes, call quality review, and voice search. Accepts ogg, mp3, and wav uploads; transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":1.25,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"above_list","discount_pct":-8.33,"list_currency":"CNY","list_input":0.9,"list_input_usd":0.13846153846153847,"list_fx_rate":6.5,"selling_usd":0.15,"unit":"asr","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"step-asr-1.1-stream","login_required":0,"quota_type":0,"model_ratio":3.3333333,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"CNY","list_input":2.6,"list_input_usd":0.4,"list_fx_rate":6.5,"selling_usd":0.39999999599999997,"unit":"asr","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-09"}},{"model_name":"doubao-seedance-2-0-mini-260615","description":"ByteDance Seedance 2.0 Mini — the lightweight, budget-friendly tier of the Seedance 2.0 line, capped at 480p and 720p, 24 fps, 4-15 seconds. It carries the same documented feature set as the rest of the line: text-to-video, first-frame and first-and-last-frame image-to-video, multimodal reference generation from 1-9 images and up to 3 reference videos totalling 15 seconds, video editing, video extension, sound-on generation, and a web-search tool; an audio reference has to accompany an image or video. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,Fast","vendor_id":5,"quota_type":0,"model_ratio":1.75,"model_price":0,"owner_by":"","completion_ratio":1,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":3.5,"list_input_usd":3.5,"selling_usd":3.5,"unit":"text","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-08-06"}},{"model_name":"glm-5.2","description":"Zhipu GLM-5.2 — flagship foundation model specialised for long-horizon coding-agent work through months of dedicated training rather than a context extension alone, with a 1M-token context window and up to 128K output tokens. Text in, text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":2,"quota_type":0,"model_ratio":0.5599999999999999,"model_price":0,"owner_by":"","completion_ratio":3.142857,"cache_ratio":0.185714,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":20,"list_currency":"USD","list_input":1.4,"list_input_usd":1.4,"selling_usd":1.1199999999999999,"unit":"text","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-07-30"}},{"model_name":"image-01","description":"MiniMax image-01 — text-to-image generation over MiniMax's native image_generation endpoint. Billed per generated image. Listed by the vendor under legacy models.","login_required":0,"tags":"Image","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.0035,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.0035,"list_input_usd":0.0035,"selling_usd":0.0035,"unit":"per_call","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-08-06"}},{"model_name":"mimo-v2.5-pro","description":"Xiaomi MiMo-V2.5-Pro — trillion-parameter flagship (42B active) with 1M context and 128K max output, tuned for high-intensity agent workloads. Supports deep thinking, function calling, structured output, and web search. Text generation only: unlike mimo-v2.5, the vendor's capability list does not include full-modality understanding.","login_required":0,"tags":"Reasoning,Tools,Files,1M","vendor_id":16,"quota_type":0,"model_ratio":0.2175,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.00827586,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.435,"list_input_usd":0.435,"selling_usd":0.435,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-07-31"}},{"model_name":"MiniMax-M2.7","description":"MiniMax M2.7 — text model for chat, coding, office work, and agentic tasks, with a 204,800-token context window and output at roughly 60 tokens per second. Text only: the API accepts text and tool-call content blocks, not images or video. Supports tool calling; thinking is always on and cannot be disabled.","specs":"{\"type\":\"llm\",\"context_window\":204800,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.143836,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":4.11,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.287672,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-07-30"}},{"model_name":"doubao-seedance-2-0-260128","description":"ByteDance Seedance 2.0 — the flagship of the Seedance 2.0 line and its only member that reaches 1080p and 4K (10-bit depth) on top of 480p and 720p, at 24 fps and 4-15 seconds. Covers text-to-video, first-frame and first-and-last-frame image-to-video, multimodal reference generation from 1-9 images and up to 3 reference videos totalling 15 seconds, video editing, video extension, sound-on generation, and a web-search tool; an audio reference has to accompany an image or video. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16, with the closing frame available as an image. Billed by output resolution.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,4K","vendor_id":5,"quota_type":0,"model_ratio":3.5,"model_price":0,"owner_by":"","completion_ratio":1,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":7,"list_input_usd":7,"selling_usd":7,"unit":"text","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-08-06"}},{"model_name":"qwen3-asr-flash","description":"Alibaba Qwen3-ASR-Flash — speech recognition built on the Qwen3 foundation model, which automatically determines the spoken language and transcribes 11 languages; the Qwen3-ASR line documents Mandarin alongside Sichuanese, Min Nan, Wu, and Cantonese, and multiple English accents. Accepts aac, amr, avi, aiff, flac, flv, mkv, mp3, mpeg, ogg, opus, wav, webm, wma, and wmv up to 5 minutes and 10 MB per request; pcm input must be 16 kHz and other formats are resampled to 16 kHz before recognition.","login_required":0,"tags":"Audio,ASR,Multilingual,Low-latency","vendor_id":4,"quota_type":0,"model_ratio":1.029488,"model_price":0,"owner_by":"","audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"below_list","discount_pct":1.95,"list_currency":"USD","list_input":0.126,"list_input_usd":0.126,"selling_usd":0.12353855999999999,"unit":"asr","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-08-03"}},{"model_name":"qwen3.6-flash","description":"Alibaba Qwen3.6-Flash — fast vision-language model that the vendor highlights for agentic coding and for spatial intelligence, with marked gains in object localisation and detection. 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 65,536 output tokens, and a 131,072-token chain-of-thought budget. Accepts text, image, and video input and returns text. Supports function calling and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.092307,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":26.15,"list_currency":"USD","list_input":0.25,"list_input_usd":0.25,"selling_usd":0.184614,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-07-31"}},{"model_name":"step-tts-2","description":"StepFun step-tts-2 — end-to-end NTP-based text-to-speech built for accurate accent reproduction. Speaks Chinese, English, mixed Chinese-English, and Japanese, plus Cantonese and Sichuanese. Emotion and delivery are directed through natural-language labels, and zero-shot voice cloning takes roughly 10 seconds of reference audio, with the emotion and style controls carrying over to the cloned voice at no extra cost. Outputs wav, mp3, flac, opus, or pcm.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice,VoiceClone","vendor_id":18,"quota_type":0,"model_ratio":20,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.4,"list_input_usd":0.4,"selling_usd":0.4,"unit":"tts","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"stepaudio-2.5-asr-stream","login_required":0,"quota_type":0,"model_ratio":1.5,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.18,"list_input_usd":0.18,"selling_usd":0.18,"unit":"asr","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-09"}},{"model_name":"wan2.7-t2v","description":"Alibaba Wan2.7 (T2V) — text-to-video with upgraded acting, nuanced emotion, intense action, and dramatic camera cuts. Produces 720P or 1080P H.264 MP4 of 2-15 seconds (5s default) in 16:9, 9:16, 1:1, 4:3, or 3:4, from prompts up to 5000 characters. Audio is included by default: with no audio_url the model composes matching background music or sound effects, or a 2-30 second wav or mp3 track can be supplied instead.","login_required":0,"tags":"Video,Text-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.092308,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.6,"list_input_usd":0.0923076923076923,"list_fx_rate":6.5,"selling_usd":0.092308,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/text-to-video-api-reference","verified_at":"2026-07-31"}},{"model_name":"agnes-video-v2.0","description":"Agnes AI Video V2.0 — an asynchronous video generation model for text-to-video, image-to-video, multi-image keyframes, and cinematic motion, supporting normalized 480p, 720p, and 1080p output tiers.","specs":"{\"type\":\"video\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"video\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Video,Text-to-Video,Image-to-Video","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second"},{"model_name":"glm-5.1","description":"Zhipu GLM-5.1 — flagship foundation model built for extended autonomous work: the vendor documents continuous unattended execution on a single task for up to 8 hours across planning, execution, testing, and iterative optimization. 200K context window, up to 128K output tokens, text in and text out. Supports multiple thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":200000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,200K","vendor_id":2,"quota_type":0,"model_ratio":0.5599999999999999,"model_price":0,"owner_by":"","completion_ratio":3.142857,"cache_ratio":0.185714,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":20,"list_currency":"USD","list_input":1.4,"list_input_usd":1.4,"selling_usd":1.1199999999999999,"unit":"text","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-07-30"}},{"model_name":"happyhorse-1.1-t2v","description":"Alibaba HappyHorse 1.1 (T2V) — text-to-video with improved semantic understanding, camera control, and dynamic generation. Fluid, detailed, physically consistent 720P/1080P video.","login_required":0,"tags":"Video,Text-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.083077,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.54,"list_input_usd":0.08307692307692308,"list_fx_rate":6.5,"selling_usd":0.083077,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-05"}},{"model_name":"kimi-k3","description":"Moonshot Kimi K3 — 2.8-trillion-parameter flagship for long-horizon coding, end-to-end knowledge work, and deep reasoning, with a 1M-token context window and up to 1,048,576 completion tokens (131,072 by default). Accepts text, base64 images, and video and returns text. Thinking is always active with reasoning_effort selectable as low, high, or max (default max); temperature is fixed at 1.0. Supports custom tool calling and dynamic tool loading.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":1048576,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"low\",\"high\",\"max\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,1M","vendor_id":3,"quota_type":0,"model_ratio":1.36986301,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":8.68,"list_currency":"USD","list_input":3,"list_input_usd":3,"selling_usd":2.73972602,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat-k3","verified_at":"2026-08-06"}},{"model_name":"kling-v3","description":"Kuaishou Kling 3.0 — text-to-video and image-to-video with native audio and improved element consistency. Clips run 3-15 seconds (5 by default) in 16:9, 9:16, or 1:1, with the tier chosen through mode: std for 720P, pro for 1080P, and 4k for native 4K. Audio is opt-in through sound=on and defaults to off; image-to-video takes a first frame with an optional tail frame. From $0.083/s (720p).","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Audio,4K","vendor_id":6,"quota_type":1,"model_ratio":0,"model_price":0.42,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.42,"list_input_usd":0.42,"selling_usd":0.42,"unit":"per_call","list_source":"https://kling.ai/dev/pricing","verified_at":"2026-07-31"}},{"model_name":"mimo-v2.5-tts-voicedesign","description":"Xiaomi MiMo voice design model: describe the desired voice in natural language via instructions and it synthesises speech in that designed voice.","login_required":0,"tags":"Audio,TTS,VoiceDesign","vendor_id":16,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-design"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters"},{"model_name":"glm-5v-turbo","description":"Zhipu GLM-5V-Turbo — multimodal reasoning model for images, video, files, coding, and tool-driven agent workflows.","login_required":0,"tags":"Reasoning,Tools,Vision,Video,Files,200K","vendor_id":2,"quota_type":0,"model_ratio":0.384615,"model_price":0,"owner_by":"","completion_ratio":4.4,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":35.9,"list_currency":"USD","list_input":1.2,"list_input_usd":1.2,"selling_usd":0.76923,"unit":"text","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-08-03"}},{"model_name":"mimo-v2.5-tts-voiceclone","description":"Xiaomi MiMo voice cloning model: supply a reference audio sample as a data URL in the voice field (data:{mime};base64,...) and it synthesises speech in that voice with no training, annotation, or fine-tuning step. Accepts MP3 (audio/mpeg) or WAV, with the base64-encoded string capped at 10 MB; Xiaomi publishes no minimum reference duration. Natural-language style instructions and inline audio tags carry over from mimo-v2.5-tts, but streaming is not available — the request runs in compatibility mode and returns once synthesis has completed.","login_required":0,"tags":"Audio,TTS,VoiceClone","vendor_id":16,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech","voice-cloning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters"},{"model_name":"MiniMax-Hailuo-02","description":"MiniMax Hailuo 02 — budget video generation covering both text-to-video and image-to-video at 24 fps, with 512p and 768p available as 6-second or 10-second clips and 1080p at 6 seconds. The 512p tier is the entry point of the Hailuo line.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.28,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.28,"list_input_usd":0.28,"selling_usd":0.28,"unit":"per_call","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-08-06"}},{"model_name":"step-1o-audio","description":"StepFun step-1o-audio — earlier-generation end-to-end voice model with multi-language support and tool calling, capped at 30 minutes per interaction. Largely superseded by the StepAudio 2.5 line; retained for workloads already built against it.","login_required":0,"tags":"Audio,Tools,Multilingual","vendor_id":18,"quota_type":0,"model_ratio":1.925,"model_price":0,"owner_by":"","completion_ratio":2.4,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"above_list","discount_pct":-0.1,"list_currency":"CNY","list_input":25,"list_input_usd":3.8461538461538463,"list_fx_rate":6.5,"selling_usd":3.85,"unit":"text","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"step-3.5-flash","description":"StepFun step-3.5-flash — text-only reasoning model with 256K context, aimed at logic, mathematics, software engineering, and deep-research work where reasoning quality has to come with fast, reliable execution. Reasoning depth is selectable per request through reasoning_effort (low / medium / high). Supports tool calling, JSON Schema structured output, streaming, and prompt caching. For image or video input use step-3.7-flash.","login_required":0,"tags":"Reasoning,Tools,256K","vendor_id":18,"quota_type":0,"model_ratio":0.05,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.1,"list_input_usd":0.1,"selling_usd":0.1,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"step-3.7-flash","description":"StepFun step-3.7-flash — sparse MoE with 198B total and 11B active parameters, native 256K context, and native understanding of images and video alongside text. Reasoning depth is selectable per request through reasoning_effort (low / medium / high), and the model supports reliable multi-step tool calling, JSON Schema structured output, streaming, and prompt caching. Tuned for agent and coding workloads.","login_required":0,"tags":"Vision,Video,Reasoning,Tools,256K","vendor_id":18,"quota_type":0,"model_ratio":0.1,"model_price":0,"owner_by":"","completion_ratio":5.75,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai","anthropic","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.2,"list_input_usd":0.2,"selling_usd":0.2,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"step-image-edit-2","description":"StepFun Step Image Edit 2 — unified text-to-image and image-editing model under 6B parameters that matches open-weight editors in the 12B-20B range, completing an edit in one to two seconds and built for low-latency interactive retouching. Supports 256x256, 512x512, 768x768, 1024x1024, 1280x800, and 800x1280, plus negative prompts and a text-optimization mode; one image per request. Served over the OpenAI-compatible images endpoint.","login_required":0,"tags":"Image,Editing,Fast","vendor_id":18,"quota_type":1,"model_ratio":0,"model_price":0.0035,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"above_list","discount_pct":-13.75,"list_currency":"CNY","list_input":0.02,"list_input_usd":0.003076923076923077,"list_fx_rate":6.5,"selling_usd":0.0035,"unit":"per_call","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"doubao-seedance-2-0-fast-260128","description":"ByteDance Seedance 2.0 Fast — cost-efficient video generation carrying the full Seedance 2.0 feature set but capped at 480p and 720p, 24 fps, 4-15 seconds. Covers text-to-video, first-frame and first-and-last-frame image-to-video, multimodal reference generation from 1-9 images and up to 3 reference videos totalling 15 seconds, video editing, video extension, sound-on generation, and a web-search tool; an audio reference has to accompany an image or video. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16. 480p 5s ≈ $0.26.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,Fast","vendor_id":5,"quota_type":0,"model_ratio":2.8,"model_price":0,"owner_by":"","completion_ratio":1,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":5.6,"list_input_usd":5.6,"selling_usd":5.6,"unit":"text","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-08-06"}},{"model_name":"qwen3-tts-flash","description":"Alibaba Qwen3-TTS-Flash — low-latency text-to-speech with 17 expressive built-in voices, covering Chinese, English, German, Italian, Portuguese, Spanish, Japanese, Korean, French, and Russian plus an automatic mode for mixed-language text, and dialect output. Accepts up to 512 tokens of text per request and supports both streaming and non-streaming synthesis.","login_required":0,"tags":"Audio,TTS,Multilingual,Low-latency","vendor_id":4,"quota_type":0,"model_ratio":6.153847,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.8,"list_input_usd":0.12307692307692308,"list_fx_rate":6.5,"selling_usd":0.12307694,"unit":"tts","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-03"}},{"model_name":"gemini-3.6-flash","description":"Google Gemini 3.6 Flash — a stable multimodal reasoning model that balances speed and intelligence for agentic and real-world tasks, with a 1,048,576-token input context and 65,536-token output limit. It accepts text, images, video, audio, and PDFs, and supports thinking, function calling, code execution, search grounding, structured output, and Computer Use (preview).","specs":"{\"type\":\"llm\",\"context_window\":1048576,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\",\"audio\",\"pdf\"],\"output_modalities\":[\"text\"],\"reasoning_efforts\":[\"medium\",\"high\"]}","login_required":0,"icon":"Gemini.Color","tags":"Reasoning,Tools,Vision,Multimodal,1M,Fast,Files","vendor_id":21,"quota_type":0,"model_ratio":0.75,"model_price":0,"owner_by":"","completion_ratio":5,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_mode":"tiered_expr","billing_expr":"tier(\"standard\", p * 1.5 + c * 7.5 + cr * 0.15 + ai * 1.5)","billing_rate":1,"billing_rate_source":"vendor","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.5,"list_input_usd":1.5,"selling_usd":1.5,"unit":"text","list_source":"https://ai.google.dev/gemini-api/docs/pricing","verified_at":"2026-08-10"}},{"model_name":"step-audio-r1.5","description":"StepFun step-audio-r1.5 — end-to-end speech model in the same family as step-audio-2, sharing its input rate while billing output 50% higher. StepFun publishes no separate capability page for this model; consult the platform audio documentation for the current parameter set.","login_required":0,"tags":"Audio","vendor_id":18,"quota_type":0,"model_ratio":0.775,"model_price":0,"owner_by":"","completion_ratio":10.5,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"above_list","discount_pct":-0.75,"list_currency":"CNY","list_input":10,"list_input_usd":1.5384615384615385,"list_fx_rate":6.5,"selling_usd":1.55,"unit":"text","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"stepaudio-2.5-chat","description":"StepFun StepAudio 2.5 Chat — end-to-end speech understanding model: it takes audio or text in and returns text, reading paralinguistic signal such as hesitation and laughter from the waveform itself rather than from a transcript. Supports extensive persona customization covering character, speech habits, and emotional range. Audio is supplied as input_audio content parts on the chat completions endpoint. For spoken replies use a TTS model.","login_required":0,"tags":"Audio,Reasoning,Tools","vendor_id":18,"quota_type":0,"model_ratio":0.75,"model_price":0,"owner_by":"","completion_ratio":2.333333,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal","audio-speech"],"tasks":["chat","reasoning","audio-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.5,"list_input_usd":1.5,"selling_usd":1.5,"unit":"text","list_source":"https://platform.stepfun.ai/docs/en/guides/pricing/details","verified_at":"2026-08-04"}},{"model_name":"wan2.7-r2v","description":"Alibaba Wan2.7 (R2V) — reference-to-video, up to 5 mixed image/video refs + audio-timbre reference, stronger character/prop/scene consistency.","login_required":0,"tags":"Video,Reference-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.092308,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["reference-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.6,"list_input_usd":0.0923076923076923,"list_fx_rate":6.5,"selling_usd":0.092308,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-05"}},{"model_name":"doubao-seed-2-1-pro-260628","description":"ByteDance Doubao Seed 2.1 Pro — flagship Doubao agent model for production work, covering deep thinking, text generation, multimodal understanding, tool calling, and structured output, for which the vendor recommends json_schema mode. 256K context window with 256K maximum input, up to 256K output tokens (4K by default), and a 256K chain-of-thought budget. Accepts text, image, and video and returns text; audio understanding is not listed for the 2.1 line. Supports implicit and explicit context caching, including prefix and session caches.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":256000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Multimodal,256K","vendor_id":5,"quota_type":0,"model_ratio":0.461539,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":6,"list_input_usd":0.9230769230769231,"list_fx_rate":6.5,"selling_usd":0.923078,"unit":"text","list_source":"https://www.volcengine.com/docs/82379","verified_at":"2026-07-31"}},{"model_name":"mimo-v2.5","description":"Xiaomi MiMo-V2.5 — natively full-modality model with 1M context and 128K max output, understanding images, video, audio, and text in one model. Supports deep thinking, function calling, structured output, and web search, with full-modality agent capability.","login_required":0,"tags":"Reasoning,Tools,Files,Vision,Video,1M","vendor_id":16,"quota_type":0,"model_ratio":0.07,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"text","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-07-31"}},{"model_name":"MiniMax-Hailuo-2.3-Fast","description":"MiniMax Hailuo 2.3 Fast — the speed-and-cost tier of Hailuo 2.3, focused on image-to-video and on physical motion realism, at 24 fps, with 768p available as 6-second or 10-second clips and 1080p at 6 seconds. 768p 6s = $0.19.","login_required":0,"tags":"Video,Image-to-Video,Fast","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.19,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.19,"list_input_usd":0.19,"selling_usd":0.19,"unit":"per_call","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-08-06"}},{"model_name":"MiniMax-M3","description":"MiniMax M3 — frontier multimodal coding model for agentic reasoning, tool use, and structured task execution, with a 1,000,000-token context window. Accepts text, images up to 10 MB, and video up to 50 MB inline or 512 MB through the Files API, and returns text. Supports tool calling, with thinking optional rather than always on.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,1M","vendor_id":13,"quota_type":0,"model_ratio":0.1438355,"model_price":0,"owner_by":"","completion_ratio":4.00000347619329,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_mode":"tiered_expr","billing_expr":"param(\"service_tier\") == \"priority\" ? (len \u003c= 512000 ? tier(\"priority\", p * 0.431507 + c * 1.726027 + cr * 0.086301) : tier(\"priority_long_context\", p * 0.863014 + c * 3.452055 + cr * 0.172603)) : (len \u003c= 512000 ? tier(\"standard\", p * 0.287671 + c * 1.150685 + cr * 0.057534) : tier(\"long_context\", p * 0.575342 + c * 2.30137 + cr * 0.115068))","billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":4.11,"list_currency":"USD","list_input":0.3,"list_input_usd":0.3,"selling_usd":0.287671,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-08-03"}},{"model_name":"qwen3.6-plus","description":"Alibaba Qwen3.6-Plus — balanced model for general reasoning and tool use, with a 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 65,536 output tokens, and an 81,920-token chain-of-thought budget. Accepts text, image, and video input and returns text. Supports function calling, structured outputs, web search, prefix completion, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.2,"model_price":0,"owner_by":"","completion_ratio":6,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":20,"list_currency":"USD","list_input":0.5,"list_input_usd":0.5,"selling_usd":0.4,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-07-31"}},{"model_name":"wan2.7-image-pro","description":"Alibaba Wan 2.7 Image Pro — the professional tier of the Wan 2.7 image line, covering text-to-image, sequential image sets, image editing, and multi-image reference generation with improved prompt following. Text-to-image reaches 4K (4096x4096) with 1K and 2K tiers and custom sizes from 768x768; editing and sequential modes cap at 2K. Accepts up to 9 reference images (JPEG, JPG, PNG, BMP, WEBP; 240-8000 px per side, 20 MB each), aspect ratios from 1:8 to 8:1, and prompts up to 5000 characters. Returns PNG.","login_required":0,"tags":"Image,4K","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.06,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"below_list","discount_pct":22,"list_currency":"CNY","list_input":0.5,"list_input_usd":0.07692307692307693,"list_fx_rate":6.5,"selling_usd":0.06,"unit":"per_call","list_source":"https://www.alibabacloud.com/help/en/model-studio/wan-image-generation-and-editing-api-reference","verified_at":"2026-07-31"}},{"model_name":"doubao-seedream-4-5-251128","description":"ByteDance Doubao Seedream 4.5 — text-to-image and image-to-image with multi-image fusion, producing either a single image or a set of related images. Single-image mode takes a prompt alone, one reference image, or 2-14 reference images; group mode (sequential_image_generation=auto) returns up to 15 images from text, up to 14 from a single reference image, or a set from 2-14 references where inputs plus outputs stay at or below 15. Supports 2K and 4K output and returns JPEG by default, as a URL or base64. Interactive coordinate-based editing is exclusive to Seedream 5.0 pro.","login_required":0,"tags":"Image,4K","vendor_id":5,"quota_type":1,"model_ratio":0,"model_price":0.038462,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.25,"list_input_usd":0.038461538461538464,"list_fx_rate":6.5,"selling_usd":0.038462,"unit":"per_call","list_source":"https://ai.volcengine.com/model","verified_at":"2026-08-10"}},{"model_name":"happyhorse-1.1-r2v","description":"Alibaba HappyHorse 1.1 (R2V) — reference-to-video with up to 9 reference images, strong subject/scene/style consistency and camera control. 720P/1080P.","login_required":0,"tags":"Video,Reference-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.083077,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["reference-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.54,"list_input_usd":0.08307692307692308,"list_fx_rate":6.5,"selling_usd":0.083077,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-05"}},{"model_name":"mimo-v2.5-asr","description":"Xiaomi MiMo speech recognition model optimized for Chinese and English plus Chinese dialects, handling noisy, far-field, and multi-speaker audio as well as lyrics transcription.","login_required":0,"tags":"Audio,ASR,Dialects,Transcription","vendor_id":16,"quota_type":0,"model_ratio":0.616667,"model_price":0,"owner_by":"","audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"USD","list_input":0.074,"list_input_usd":0.074,"selling_usd":0.07400003999999999,"unit":"asr","list_source":"https://mimo.mi.com/docs/zh-CN/price/pay-as-you-go","verified_at":"2026-07-31"}},{"model_name":"step-2x-large","description":"StepFun Step 2X Large — text-to-image generation with realistic texture and unusually strong Chinese and English text rendering inside the image itself. Supports 256x256, 512x512, 768x768, 1024x1024, 1280x800, and 800x1280; one image per request. Served over the OpenAI-compatible images endpoint.","login_required":0,"tags":"Image","vendor_id":18,"quota_type":1,"model_ratio":0,"model_price":0.0155,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"above_list","discount_pct":-0.75,"list_currency":"CNY","list_input":0.1,"list_input_usd":0.015384615384615385,"list_fx_rate":6.5,"selling_usd":0.0155,"unit":"per_call","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"deepseek-v4-flash","description":"DeepSeek V4 Flash — the fast, high-concurrency DeepSeek V4 tier for chat, coding help, and agent loops, with a 1M-token context window and a 384K maximum output. It runs in both thinking (default) and non-thinking modes and supports tool calling and JSON output. Text in, text out.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":384000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":1,"quota_type":0,"model_ratio":0.07,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.02,"enable_groups":["default"],"supported_endpoint_types":["openai","openai-response"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.14,"list_input_usd":0.14,"selling_usd":0.14,"unit":"text","list_source":"https://api-docs.deepseek.com/quick_start/pricing","verified_at":"2026-08-06"}},{"model_name":"kimi-k2.7-code-highspeed","description":"Moonshot Kimi K2.7 Code Highspeed — the throughput tier of K2.7 Code, serving the same model at roughly 180 tokens per second and up to 260 in short-context work. 256K context window, 32,768 default output tokens. Accepts text, images, and video as base64 content and returns text. Thinking is mandatory and returns an error if disabled; tool_choice is limited to auto or none and reasoning_content must be carried across multi-step tool calls.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"Moonshot.Color","tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,256K","vendor_id":3,"quota_type":0,"model_ratio":0.89041096,"model_price":0,"owner_by":"","completion_ratio":4.153846,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":6.27,"list_currency":"USD","list_input":1.9,"list_input_usd":1.9,"selling_usd":1.78082192,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat-k27-code","verified_at":"2026-08-06"}},{"model_name":"MiniMax-Hailuo-2.3","description":"MiniMax Hailuo 2.3 — text-to-video and image-to-video that the vendor positions on instruction following, at 24 fps, with 768p available as 6-second or 10-second clips and 1080p at 6 seconds. 768p 6s = $0.28.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.28,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.28,"list_input_usd":0.28,"selling_usd":0.28,"unit":"per_call","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-08-06"}},{"model_name":"MiniMax-M2.7-highspeed","description":"MiniMax M2.7 Highspeed — the same M2.7 model served at roughly 100 tokens per second for low-latency coding and agent workflows, with a 204,800-token context window. Text only: the API accepts text and tool-call content blocks, not images or video. Supports tool calling; thinking is always on and cannot be disabled.","specs":"{\"type\":\"llm\",\"context_window\":204800,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,200K","vendor_id":13,"quota_type":0,"model_ratio":0.287671,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":0.1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":4.11,"list_currency":"USD","list_input":0.6,"list_input_usd":0.6,"selling_usd":0.575342,"unit":"text","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-07-30"}},{"model_name":"speech-2.8-hd","description":"MiniMax speech-2.8-hd — the high-definition tier of the 2.8 speech line, positioned for ultra-realistic delivery with sound tags, covering 40 languages and 7 emotions. Pitch, speech rate, volume, sample rate, bitrate, and output format are configurable per request, and long-form synthesis accepts up to 1M characters asynchronously or 10,000 characters over the streaming interface.","login_required":0,"tags":"Audio,TTS,Multilingual,High-definition","vendor_id":13,"quota_type":0,"model_ratio":26.923077,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"below_list","discount_pct":46.15,"list_currency":"USD","list_input":1,"list_input_usd":1,"selling_usd":0.53846154,"unit":"tts","list_source":"https://platform.minimax.io/docs/guides/pricing","verified_at":"2026-07-31"}},{"model_name":"wan2.7-i2v","description":"Alibaba Wan 2.7 Image-to-Video — covers first-frame generation, first-and-last-frame transitions, and continuation of an existing clip. Produces 720P or 1080P (default 1080P) of 2-15 seconds, 5s default, from prompts up to 5000 characters (negative prompts up to 500). A 2-30 second mp3 or wav driving_audio track can be supplied; with no audio the model generates matching background music or sound effects, audio longer than the clip is truncated, and audio shorter than it leaves the tail silent.","login_required":0,"tags":"Video,Image-to-Video,Audio","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.092308,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["image-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.6,"list_input_usd":0.0923076923076923,"list_fx_rate":6.5,"selling_usd":0.092308,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/text-to-video-api-reference","verified_at":"2026-07-31"}},{"model_name":"agnes-2.0-flash","description":"Agnes AI Agnes 2.0 Flash — a fast multimodal agent model with a 256K input context and 64K reference output limit, supporting chat, streaming, tool calling, coding, reasoning, image understanding, and agent workflows.","specs":"{\"type\":\"llm\",\"context_window\":262144,\"max_output_tokens\":65536,\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"text\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Reasoning,Tools,Vision,Multimodal,256K,Fast","vendor_id":25,"quota_type":0,"model_ratio":0,"model_price":0,"owner_by":"","completion_ratio":0,"cache_ratio":0,"create_cache_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens"},{"model_name":"doubao-seed-2-1-turbo-260628","description":"ByteDance Doubao Seed 2.1 Turbo — the low-latency lane of the Seed 2.1 line, sharing Pro's envelope: 256K context window with 256K maximum input, up to 256K output tokens (4K by default), and a 256K chain-of-thought budget. Covers deep thinking, text generation, multimodal understanding, tool calling, and structured output, though without Pro's json_schema recommendation. Accepts text, image, and video and returns text; audio understanding is not listed for the 2.1 line. Supports implicit and explicit context caching.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":256000,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,Fast,256K","vendor_id":5,"quota_type":0,"model_ratio":0.230769,"model_price":0,"owner_by":"","completion_ratio":5,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"CNY","list_input":3,"list_input_usd":0.46153846153846156,"list_fx_rate":6.5,"selling_usd":0.461538,"unit":"text","list_source":"https://www.volcengine.com/docs/82379","verified_at":"2026-07-31"}},{"model_name":"doubao-seedream-5-0-260128","description":"ByteDance Doubao Seedream 5.0-lite — text-to-image and image-to-image with smarter controllable creation, real-time retrieval, and stronger subject consistency (2K+ resolution).","login_required":0,"tags":"Image,4K","vendor_id":5,"quota_type":1,"model_ratio":0,"model_price":0.033847,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image","high-resolution"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.22,"list_input_usd":0.033846153846153845,"list_fx_rate":6.5,"selling_usd":0.033847,"unit":"per_call","list_source":"https://www.volcengine.com/docs/82379","verified_at":"2026-07-31"}},{"model_name":"happyhorse-1.1-i2v","description":"Alibaba HappyHorse 1.1 (I2V) — image-to-video with better texture, ID consistency across clips, motion fluidity, and audio-visual sync. 720P/1080P.","login_required":0,"tags":"Video,Image-to-Video","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.083077,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["image-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.54,"list_input_usd":0.08307692307692308,"list_fx_rate":6.5,"selling_usd":0.083077,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-05"}},{"model_name":"step-1o-turbo-vision","description":"StepFun step-1o-turbo-vision — image and video understanding with fast text generation, 32K context. Accepts text, image, and video input and returns text only. An earlier-generation vision model retained for compatibility; step-3.7-flash covers the same input modalities with 256K context and selectable reasoning depth. Keep images under 4096 pixels on the long edge for reliable recognition.","login_required":0,"tags":"Vision,Video,32K","vendor_id":18,"quota_type":0,"model_ratio":0.2,"model_price":0,"owner_by":"","completion_ratio":3.2,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"above_list","discount_pct":-4,"list_currency":"CNY","list_input":2.5,"list_input_usd":0.38461538461538464,"list_fx_rate":6.5,"selling_usd":0.4,"unit":"text","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"wan2.7-videoedit","description":"Alibaba Wan2.7 VideoEdit — natural-language video editing, local or global, reference-image element replacement, motion/effect/camera replication.","login_required":0,"tags":"Video,Video-Edit","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.092308,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["video-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":0.6,"list_input_usd":0.0923076923076923,"list_fx_rate":6.5,"selling_usd":0.092308,"unit":"per_second","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-05"}},{"model_name":"agnes-image-2.1-flash","description":"Agnes AI Image 2.1 Flash — a detail-focused image generation and editing model for high-information-density scenes, text-to-image, image-to-image, and multi-image composition across flexible 1K–4K sizes and aspect ratios.","specs":"{\"type\":\"image\",\"input_modalities\":[\"text\",\"image\"],\"output_modalities\":[\"image\"]}","login_required":0,"icon":"https://dash.chinaapi.ai/providers/agnes.png","tags":"Image,Editing,High-definition","vendor_id":25,"quota_type":1,"model_ratio":0,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation"],"capabilities":["image"],"tasks":["text-to-image","image-editing"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call"},{"model_name":"doubao-seedance-2-5-260628","description":"ByteDance Seedance 2.5 — the mainline Seedance model, stretching a single generation to 4-30 seconds at 24 fps while capping resolution at 480p and 720p. Multimodal reference generation scales to 1-30 images, up to 10 reference videos, and up to 10 reference audio clips (30 seconds total each), and unlike the 2.0 line an audio reference can stand on its own. Also covers text-to-video, first-frame and first-and-last-frame image-to-video, video editing, video extension, sound-on generation, and a web-search tool. Returns MP4 in 21:9, 16:9, 4:3, 1:1, 3:4, or 9:16.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio","vendor_id":5,"quota_type":0,"model_ratio":5.35,"model_price":0,"owner_by":"","completion_ratio":1,"cache_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":10.7,"list_input_usd":10.7,"selling_usd":10.7,"unit":"text","list_source":"https://docs.byteplus.com/en/docs/ModelArk/1544106","verified_at":"2026-08-06"}},{"model_name":"glm-tts","description":"Zhipu GLM-TTS — text-to-speech built on a two-stage generation architecture with GRPO reinforcement learning, which infers emotion and intonation from the surrounding text instead of requiring explicit markup. Ships seven built-in voices (tongtong by default, xiaochen, chuichui, jam, kazi, douji, luodo) with adjustable speed and volume, accepts up to 1024 characters per request, and streams with first-frame latency under 400 ms. Returns wav or pcm at a recommended 24 kHz sample rate; streaming is pcm only.","login_required":0,"tags":"Audio,TTS,Multilingual,Voice","vendor_id":2,"quota_type":0,"model_ratio":15.384615,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"CNY","list_input":2,"list_input_usd":0.3076923076923077,"list_fx_rate":6.5,"selling_usd":0.30769230000000003,"unit":"tts","list_source":"https://open.bigmodel.cn/pricing","verified_at":"2026-08-09"}},{"model_name":"kling-v3-omni","description":"Kling 3.0 Omni — reference-based multi-image video generation. Listed price is the std tier (720p, 5s, no audio, no reference video). Requests that do not set mode use the upstream default pro tier, which bills 1.33x — about $0.56 for the same 5s clip; 4k bills 5x. Pass mode=std to be charged the listed price.","login_required":0,"tags":"Video,Reference-to-Video","vendor_id":6,"quota_type":1,"model_ratio":0,"model_price":0.42,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["reference-to-video"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.42,"list_input_usd":0.42,"selling_usd":0.42,"unit":"per_call","list_source":"https://kling.ai/dev/pricing","verified_at":"2026-08-06"}},{"model_name":"MiniMax-H3","description":"MiniMax H3 (Hailuo 3.0) — next-generation open general-purpose multimodal video model producing native stereo audio in one pass, at 768P or 2K, 4-15 seconds (integers only), 24 fps. Covers text-to-video, image-to-video from a first and/or last frame, and reference generation from images, videos, or audio for character, motion, camera, style, voice, or editing rhythm; image-to-video and reference generation cannot be combined in one request. Accepts H.264/H.265 video, JPG, JPEG, PNG, WEBP, HEIC, and HEIF images, and WAV or MP3 audio, with prompts up to 7000 characters. $0.08/s at 768P, $0.13/s at 2K.","login_required":0,"tags":"Video,Text-to-Video,Image-to-Video,Reference-to-Video,Audio,Open Weights","vendor_id":13,"quota_type":1,"model_ratio":0,"model_price":0.08,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["openai-video","openai"],"capabilities":["video"],"tasks":["text-to-video","image-to-video","reference-to-video","video-with-audio"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_second","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.08,"list_input_usd":0.08,"selling_usd":0.08,"unit":"per_second","list_source":"https://platform.minimax.io/docs/guides/pricing-paygo","verified_at":"2026-08-04"}},{"model_name":"step-asr-1.1","description":"StepFun step-asr-1.1 — the updated general-purpose recognizer in the step-asr line, covering Chinese, English, and Chinese dialects. Accepts ogg, mp3, and wav uploads; transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":2.916667,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"above_list","discount_pct":-3.41,"list_currency":"CNY","list_input":2.2,"list_input_usd":0.3384615384615385,"list_fx_rate":6.5,"selling_usd":0.35000003999999996,"unit":"asr","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"deepseek-v4-pro","description":"DeepSeek V4 Pro — the higher-capability DeepSeek V4 tier for coding, reasoning, and agentic work, with a 1M-token context window and a 384K maximum output. It runs in both thinking (default) and non-thinking modes and supports tool calling and JSON output. Text in, text out.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":384000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Open Weights,1M","vendor_id":1,"quota_type":0,"model_ratio":0.2175,"model_price":0,"owner_by":"","completion_ratio":2,"cache_ratio":0.008333,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.435,"list_input_usd":0.435,"selling_usd":0.435,"unit":"text","list_source":"https://api-docs.deepseek.com/quick_start/pricing","verified_at":"2026-08-06"}},{"model_name":"glm-5-turbo","description":"Zhipu GLM-5-Turbo — the throughput-oriented GLM-5 variant, optimised for agent workflows: more stable tool and skill invocation across multi-step tasks, better handling of complex long-chain instructions, and scheduled or long-running task execution. 200K context window, up to 128K output tokens, text in and text out. Supports thinking modes, function calling, structured JSON output, streaming, context caching, and MCP tool integration.","specs":"{\"type\":\"llm\",\"context_window\":200000,\"max_output_tokens\":128000,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,200K","vendor_id":2,"quota_type":0,"model_ratio":0.6,"model_price":0,"owner_by":"","completion_ratio":3.333333,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":1.2,"list_input_usd":1.2,"selling_usd":1.2,"unit":"text","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-07-30"}},{"model_name":"glm-asr-2512","description":"Zhipu GLM-ASR-2512 — speech recognition model reporting a 0.0717 character error rate. Covers Mandarin together with Sichuanese, Cantonese, Min Nan, and Wu, American and British English accents, and dozens of further languages including French, German, Japanese, Korean, Spanish, and Arabic. Handles mixed-language speech, technical jargon, and colloquial delivery, and accepts a custom dictionary for domain vocabulary. Audio is capped at 30 seconds and 25 MB per request, with both batch and streaming transcription.","login_required":0,"tags":"Audio,ASR,Multilingual,Transcription","vendor_id":2,"quota_type":0,"model_ratio":1.2,"model_price":0,"owner_by":"","audio_ratio":1,"enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":0.144,"list_input_usd":0.144,"selling_usd":0.144,"unit":"asr","list_source":"https://docs.z.ai/guides/overview/pricing","verified_at":"2026-08-10"}},{"model_name":"kimi-k2.7-code","description":"Moonshot Kimi K2.7 Code — Kimi's dedicated coding model, which the vendor measures as improving instruction compliance and long-horizon coding over K2.6 while cutting overthinking by about 30% on average. 256K context window, 32,768 default output tokens. Accepts text, images, and video as base64 content and returns text. Thinking is mandatory and returns an error if disabled; tool_choice is limited to auto or none and reasoning_content must be carried across multi-step tool calls.","specs":"{\"type\":\"llm\",\"context_window\":256000,\"max_output_tokens\":32768,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Open Weights,Vision,Video,256K","vendor_id":3,"quota_type":0,"model_ratio":0.38,"model_price":0,"owner_by":"","completion_ratio":4.153846,"cache_ratio":0.2,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":20,"list_currency":"USD","list_input":0.95,"list_input_usd":0.95,"selling_usd":0.76,"unit":"text","list_source":"https://platform.kimi.ai/docs/pricing/chat-k27-code","verified_at":"2026-08-06"}},{"model_name":"qwen3.7-max","description":"Alibaba Qwen3.7-Max — closed-source flagship positioned for programming, office and productivity work, and long-term autonomous execution, with a 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 131,072 output tokens, and a 262,144-token chain-of-thought budget. Text in, text out. Supports function calling, structured outputs, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":131072,\"input_modalities\":[\"text\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,1M","vendor_id":4,"quota_type":0,"model_ratio":1.25,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"USD","list_input":2.5,"list_input_usd":2.5,"selling_usd":2.5,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-07-31"}},{"model_name":"qwen3.7-plus","description":"Alibaba Qwen3.7-Plus — multimodal hybrid agent model that perceives real-world scenes, reads screens and drives GUIs, generates code from visual references, and performs end-to-end navigation inside mobile apps. 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 131,072 output tokens, and a 262,144-token chain-of-thought budget. Accepts text, image, and video input and returns text; supports function calling, structured outputs, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":131072,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Files,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.153846,"model_price":0,"owner_by":"","completion_ratio":4,"cache_ratio":1,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision","file-video-understanding"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":23.08,"list_currency":"USD","list_input":0.4,"list_input_usd":0.4,"selling_usd":0.307692,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-07-31"}},{"model_name":"qwen3.8-max","description":"Alibaba Qwen3.8-Max — 2.4-trillion-parameter MoE flagship and the most capable model in the Qwen family, with the vendor citing major gains in coding and office productivity. 1,000,000-token context window (991,808 max input, 983,616 in thinking mode), up to 131,072 output tokens, and a 262,144-token chain-of-thought budget. Accepts text, image, and video input and returns text; supports function calling, structured outputs, and context caching.","specs":"{\"type\":\"llm\",\"context_window\":1000000,\"max_output_tokens\":131072,\"input_modalities\":[\"text\",\"image\",\"video\"],\"output_modalities\":[\"text\"]}","login_required":0,"tags":"Reasoning,Tools,Vision,1M","vendor_id":4,"quota_type":0,"model_ratio":0.995,"model_price":0,"owner_by":"","completion_ratio":3,"cache_ratio":0.125,"create_cache_ratio":1.25,"enable_groups":["default"],"supported_endpoint_types":["openai"],"capabilities":["text-multimodal"],"tasks":["chat","reasoning","vision"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"tokens","vendor_list":{"verdict":"below_list","discount_pct":0.5,"list_currency":"USD","list_input":2,"list_input_usd":2,"selling_usd":1.99,"unit":"text","list_source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","verified_at":"2026-08-04"}},{"model_name":"speech-2.8-turbo","description":"MiniMax speech-2.8-turbo — the low-latency tier of the 2.8 speech line, trading the HD model's ultra-realistic rendering for faster synthesis with natural flow. Covers the same 40 languages and 7 emotions, with configurable pitch, speech rate, volume, sample rate, bitrate, and output format, and accepts up to 1M characters asynchronously or 10,000 characters over the streaming interface.","login_required":0,"tags":"Audio,TTS,Multilingual,Low-latency","vendor_id":13,"quota_type":0,"model_ratio":15.384616,"model_price":0,"owner_by":"","audio_completion_ratio":0,"enable_groups":["default"],"supported_endpoint_types":["audio-speech","openai"],"capabilities":["audio-speech"],"tasks":["text-to-speech"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"characters","vendor_list":{"verdict":"at_list","discount_pct":-0,"list_currency":"CNY","list_input":2,"list_input_usd":0.3076923076923077,"list_fx_rate":6.5,"selling_usd":0.30769232,"unit":"tts","list_source":"https://platform.minimaxi.com/docs/guides/pricing-speech","verified_at":"2026-07-31"}},{"model_name":"stepaudio-2-asr-pro","description":"StepFun StepAudio 2 ASR Pro — 32B-parameter speech recognition model, the accuracy-first option in the StepFun ASR line where stepaudio-2.5-asr is the speed-and-cost option. Accepts ogg, mp3, and wav uploads; transcript output is not billed.","login_required":0,"tags":"Audio,ASR,Transcription","vendor_id":18,"quota_type":0,"model_ratio":2.916667,"model_price":0,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["audio-transcription","openai"],"capabilities":["audio-speech"],"tasks":["transcription"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"audio_duration","vendor_list":{"verdict":"above_list","discount_pct":-13.75,"list_currency":"CNY","list_input":2,"list_input_usd":0.3076923076923077,"list_fx_rate":6.5,"selling_usd":0.35000003999999996,"unit":"asr","list_source":"https://platform.stepfun.com/docs/zh/guides/pricing/details","verified_at":"2026-08-04","premium_reason":"Domestic-only model: StepFun's international site does not sell it, so the CNY list converted at 6.5 is the basis. The published unit price is rounded up to a legible figure, which is the declared premium."}},{"model_name":"wan2.7-image","description":"Alibaba Wan2.7 Image — text-to-image, image editing, multi-image reference generation, interactive editing, strong text rendering and instruction following.","login_required":0,"tags":"Image","vendor_id":4,"quota_type":1,"model_ratio":0,"model_price":0.030769,"owner_by":"","enable_groups":["default"],"supported_endpoint_types":["image-generation","openai"],"capabilities":["image"],"tasks":["text-to-image"],"billing_rate":1,"billing_rate_source":"default","billing_unit":"per_call","vendor_list":{"verdict":"at_list","discount_pct":0,"list_currency":"CNY","list_input":0.2,"list_input_usd":0.03076923076923077,"list_fx_rate":6.5,"selling_usd":0.030769,"unit":"per_call","list_source":"https://help.aliyun.com/zh/model-studio/model-pricing","verified_at":"2026-08-03"}}],"group_ratio":{"default":1},"pricing_version":"94d74add6a1a332fe6cda0102ea096de16457e31c3a667e2f5e355f84fc15138","success":true,"supported_endpoint":{"anthropic":{"path":"/v1/messages","method":"POST"},"audio-speech":{"path":"/v1/audio/speech","method":"POST"},"audio-transcription":{"path":"/v1/audio/transcriptions","method":"POST"},"image-generation":{"path":"/v1/images/generations","method":"POST"},"openai":{"path":"/v1/chat/completions","method":"POST"},"openai-response":{"path":"/v1/responses","method":"POST"},"openai-video":{"path":"/v1/videos","method":"POST"}},"usable_group":{"default":"Default"},"vendors":[{"id":2,"name":"智谱","description":"Zhipu AI — the GLM model family.","icon":"Zhipu.Color"},{"id":13,"name":"MiniMax","description":"MiniMax — MiniMax LLMs and Hailuo video generation.","icon":"Minimax.Color"},{"id":19,"name":"OpenAI","description":"OpenAI models","icon":"OpenAI"},{"id":21,"name":"Google","description":"Gemini models","icon":"Gemini.Color"},{"id":25,"name":"Agnes AI","description":"Agnes AI — multimodal text, image, and video generation models.","icon":"https://dash.chinaapi.ai/providers/agnes.png"},{"id":1,"name":"DeepSeek","description":"DeepSeek — open-weights frontier LLMs for reasoning and coding.","icon":"DeepSeek.Color"},{"id":4,"name":"阿里巴巴","description":"Alibaba — the Qwen model family.","icon":"Qwen.Color"},{"id":6,"name":"快手","description":"Kuaishou — the Kling video generation family.","icon":"Kling.Color"},{"id":15,"name":"Meituan","description":"Meituan — LongCat model family.","icon":"https://chinaapi.ai/assets/logos/longcat.png"},{"id":16,"name":"Xiaomi","description":"Xiaomi — MiMo model family.","icon":"https://chinaapi.ai/assets/logos/xiaomi.png"},{"id":17,"name":"Tencent","description":"Tencent — Hunyuan (混元) model family.","icon":"Hunyuan.Color"},{"id":3,"name":"Moonshot","description":"Moonshot AI — the Kimi model family.","icon":"Moonshot"},{"id":5,"name":"字节跳动","description":"ByteDance — Doubao LLMs and Seedance video generation.","icon":"Doubao.Color"},{"id":18,"name":"StepFun","description":"StepFun (阶跃星辰) — Step model family.","icon":"Stepfun.Color"},{"id":20,"name":"Anthropic","description":"Claude models","icon":"Claude.Color"}]}