{"object":"list","data":[{"id":"0gm-1.0-35b-a3b","object":"model","created":1788433066,"owned_by":"0G Foundation","name":"0GM-1.0-35B-A3B","description":"0G.AI in-house model optimized for agentic coding and tool use; thinking enabled by default.","type":"chatbot","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"instruct_type":"chatml","tokenizer":"qwen"},"supported_parameters":["temperature","top_p","top_k","max_tokens","frequency_penalty","presence_penalty","stop","tools","tool_choice","response_format","chat_template_kwargs","reasoning_effort"],"supported_formats":["openai","anthropic"],"default_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"pricing":{"prompt":"392000000000","completion":"2350000000000"},"pricing_usd":{"prompt":"0.00000008","completion":"0.00000048"},"verifiability":"TeeML","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1},{"id":"0gm-1.0-35b-a3b-sia","object":"model","created":1784016043,"owned_by":"0G Foundation","name":"0GM-1.0-35B-A3B-SIA","description":"A 35B hybrid MoE model enhanced with per-token reward guidance. At each decoding step, a 4B Value Model scores candidate tokens and steers generation toward higher-quality outputs, with improvements in harmlessness, helpfulness, and honesty.","type":"chatbot","context_length":32768,"max_completion_tokens":8192,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"instruct_type":"chatml","tokenizer":"qwen"},"supported_parameters":["temperature","top_p","top_k","max_tokens","frequency_penalty","presence_penalty","stop","chat_template_kwargs","reasoning_effort"],"supported_formats":["openai"],"default_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"pricing":{"prompt":"2650000000000","completion":"15930000000000"},"pricing_usd":{"prompt":"0.000000536","completion":"0.000003216"},"verifiability":"TeeML","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1},{"id":"claude-fable-5","object":"model","created":1789151944,"owned_by":"0G Foundation","name":"Claude Fable 5","description":"Anthropic Claude Fable 5; text and image input, text output, with a 1M-token context window. Extended thinking and tool use supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude"},"supported_parameters":["max_tokens","stream","system","stop_sequences","tools","tool_choice","metadata","output_config","cache_control"],"supported_formats":["anthropic"],"pricing":{"prompt":"44240000000000","completion":"221210000000000","cache_write":"55300000000000","cache_write_1h":"88480000000000"},"pricing_usd":{"prompt":"0.000009","completion":"0.000045","cache_write":"0.00001125","cache_write_1h":"0.000018"},"provider_count":1},{"id":"claude-opus-4-8","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"Claude Opus 4.8","description":"Anthropic Claude Opus 4.8; text and image input, text output, with a 1M-token context window. Extended thinking and tool use supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude"},"supported_parameters":["max_tokens","stream","system","stop_sequences","tools","tool_choice","metadata","thinking","output_config","cache_control"],"supported_formats":["anthropic"],"pricing":{"prompt":"24570000000000","completion":"122890000000000","cache_write":"30712500000000","cache_write_1h":"49140000000000"},"pricing_usd":{"prompt":"0.000005","completion":"0.000025","cache_write":"0.00000625","cache_write_1h":"0.00001"},"provider_count":2},{"id":"claude-opus-5","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"Claude Opus 5","description":"Anthropic Claude Opus 5; text and image input, text output, with a 1M-token context window. Extended thinking and tool use supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude"},"supported_parameters":["max_tokens","stream","system","tools","tool_choice","metadata","thinking","output_config","cache_control","stop_sequences"],"supported_formats":["anthropic"],"pricing":{"prompt":"24570000000000","completion":"122890000000000","cache_write":"30712500000000","cache_write_1h":"49140000000000"},"pricing_usd":{"prompt":"0.000005","completion":"0.000025","cache_write":"0.00000625","cache_write_1h":"0.00001"},"provider_count":2},{"id":"claude-sonnet-5","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"Claude Sonnet 5","description":"Anthropic Claude Sonnet 5; text and image input, text output, with a 1M-token context window. Extended thinking and tool use supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Claude"},"supported_parameters":["max_tokens","stream","system","stop_sequences","tools","tool_choice","metadata","thinking","output_config","cache_control"],"supported_formats":["anthropic"],"pricing":{"prompt":"9340000000000","completion":"46700000000000","cache_write":"11675000000000","cache_write_1h":"18680000000000"},"pricing_usd":{"prompt":"0.0000019","completion":"0.0000095","cache_write":"0.000002375","cache_write_1h":"0.0000038"},"provider_count":2},{"id":"deepseek-v4-flash","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"DeepSeek-V4-Flash","description":"Lightweight MoE (284B total / 13B active) tuned for fast, low-cost, high-throughput text work; function calling, web search, and thinking on by default (disable with enable_thinking:false); 1M context, up to 384K output. Currently served as the pinned 2026-07-31 snapshot.","type":"chatbot","context_length":1000000,"max_completion_tokens":393216,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"]},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort","reasoning","n","logprobs"],"supported_formats":["openai","anthropic"],"pricing":{"prompt":"818000000000","completion":"1630000000000"},"pricing_usd":{"prompt":"0.000000166667","completion":"0.000000333333"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"deepseek-v4-pro","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"DeepSeek-V4-Pro","description":"DeepSeek-V4-Pro for agentic coding, multi-step workflows, and complex reasoning; 1M context, up to 384K output. Currently served as the pinned 2026-08-13 snapshot.","type":"chatbot","context_length":1000000,"max_completion_tokens":393216,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"]},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","repetition_penalty","seed","stop","stream","tools","response_format","enable_thinking","reasoning_effort","frequency_penalty","logprobs","tool_choice"],"supported_formats":["openai","anthropic"],"pricing":{"prompt":"3880000000000","completion":"11660000000000"},"pricing_usd":{"prompt":"0.000000792","completion":"0.000002376"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"deepseek-v4-flash-vision-exp","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"DeepSeek-V4-Flash-Vision (experimental)","description":"DeepSeek-V4-Flash-Vision, the experimental vision-enabled build of DeepSeek-V4-Flash, served via Tencent Cloud MaaS (TokenHub). Image and text in, text out, 1M-token context, deep thinking, function calling, and implicit prompt caching.","type":"chatbot","context_length":1048576,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","presence_penalty","frequency_penalty","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"3240000000000","completion":"9720000000000"},"pricing_usd":{"prompt":"0.00000066","completion":"0.00000198"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"bytedance/seedance-2.5","object":"model","created":1786946058,"owned_by":"0G Foundation","name":"ByteDance Seedance 2.5","description":"ByteDance Seedance 2.5 text-to-video and image-to-video (single first-frame reference via input_reference) — the two Seedance capabilities that map onto a real OpenAI Video API field. ByteDance's own model also supports first+last-frame control and multimodal reference generation (multiple reference images/videos/audio composited per the prompt, including audio-only input), but OpenAI's Video API has no field to express either one, so this integration does not expose a client-facing input for them. Duration 4-30s (default 5); seconds is required on this endpoint and is clamped into that range rather than rejected (the vendor's own 'model-chosen' -1 duration is not reachable through this endpoint — seconds must be a positive number here), resolution 480p/720p/1080p (narrower than earlier Seedance versions — 4k is not served, and is downgraded rather than rejected: a 4K pixel size renders at 1080p, while the bare token '4k' is unrecognised and falls to the 720p default), ratio 16:9/9:16/4:3/3:4/1:1/21:9/adaptive, optional synchronized audio (generate_audio, on by default), fps fixed at 24, optional output_format passthrough (default mp4). Billed on the vendor-reported token count (usage.completion_tokens) — not a flat per-second rate. Async: POST /v1/videos, poll GET /v1/videos/{id} until completed, then GET /v1/videos/{id}/content for the MP4.","type":"video-generation","architecture":{"modality":"text+image-\u003evideo","input_modalities":["text","image"],"output_modalities":["video"]},"supported_parameters":["prompt","seconds","size","input_reference"],"supported_formats":["openai"],"default_parameters":{"seconds":5,"size":"720p"},"pricing":{"prompt":"0","completion":"52850000000000","video_unit":"video_token","variants":[{"dimensions":{"has_video_input":"false","resolution":"480p"},"unit":"video_token","unit_price":"52850000000000"},{"dimensions":{"has_video_input":"false","resolution":"720p"},"unit":"video_token","unit_price":"52850000000000"},{"dimensions":{"has_video_input":"false","resolution":"1080p"},"unit":"video_token","unit_price":"41608261682242"}]},"pricing_usd":{"prompt":"0","completion":"0","video_unit":"video_token","variants":[{"dimensions":{"has_video_input":"false","resolution":"480p"},"unit":"video_token","unit_price":"0.0000107"},{"dimensions":{"has_video_input":"false","resolution":"720p"},"unit":"video_token","unit_price":"0.0000107"},{"dimensions":{"has_video_input":"false","resolution":"1080p"},"unit":"video_token","unit_price":"0.000008424"}]},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1},{"id":"glm-5","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"GLM-5","description":"Zhipu AI GLM-5, purpose-built for coding and agent workflows; 744B foundation, 200K context.","type":"chatbot","context_length":202752,"max_completion_tokens":32768,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"instruct_type":"chatml","tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","frequency_penalty","presence_penalty","stop","tools","tool_choice","response_format","chat_template_kwargs","reasoning_effort","seed","reasoning"],"supported_formats":["openai","anthropic"],"default_parameters":{"temperature":0.7,"top_p":0.9},"pricing":{"prompt":"3680000000000","completion":"11780000000000"},"pricing_usd":{"prompt":"0.00000075","completion":"0.0000024"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"glm-5-turbo","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"GLM-5-Turbo","description":"Zhipu AI GLM-5-Turbo, the low-latency turbo variant of GLM-5, served via Tencent Cloud MaaS (TokenHub). Text in / text out, 195K-token context, hybrid reasoning with a deep-thinking mode, function calling, structured output, and implicit prompt caching.","type":"chatbot","context_length":195000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","presence_penalty","frequency_penalty","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"5740000000000","completion":"19150000000000"},"pricing_usd":{"prompt":"0.00000117","completion":"0.0000039"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"glm-5.1","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"GLM-5.1","description":"Zhipu AI GLM-5.1, purpose-built for long-horizon tasks; 744B foundation, 200K context.","type":"chatbot","context_length":206848,"max_completion_tokens":131072,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"instruct_type":"chatml","tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","frequency_penalty","presence_penalty","stop","tools","tool_choice","response_format","chat_template_kwargs","reasoning_effort"],"supported_formats":["openai"],"default_parameters":{"temperature":0.7,"top_p":0.9},"pricing":{"prompt":"4910000000000","completion":"19640000000000"},"pricing_usd":{"prompt":"0.000001","completion":"0.000004"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"glm-5.2","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"GLM-5.2","description":"Zhipu AI GLM-5.2, open-source, purpose-built for long-horizon tasks; 1M lossless context. Strong coding and engineering: autonomous task decomposition, architecture design, full-stack development, integration testing, and multi-platform deployment.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","frequency_penalty","presence_penalty","stop","tools","tool_choice","response_format","chat_template_kwargs","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"6540000000000","completion":"22910000000000"},"pricing_usd":{"prompt":"0.000001333333","completion":"0.000004666667"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"glm-5.2-fast-preview","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"GLM-5.2-Fast-Preview","description":"Zhipu AI GLM-5.2-Fast-Preview, the low-latency preview build of GLM-5.2 served through Alibaba Cloud Model Studio (DashScope). Text in / text out, 1M-token context, hybrid thinking/non-thinking modes, function calling, structured output, and implicit prompt caching.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"13090000000000","completion":"45830000000000"},"pricing_usd":{"prompt":"0.000002666667","completion":"0.000009333333"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"glm-5.3","object":"model","created":1788871645,"owned_by":"0G Foundation","name":"GLM-5.3","description":"Zhipu AI GLM-5.3 for long-horizon reasoning, coding, and agentic tool use; 1M context, up to 131K output. Deep thinking is always on and cannot be disabled; how reasoning depth is controlled depends on the provider (top-level reasoning_effort or chat_template_kwargs). Function calling, JSON mode (response_format: json_object), and implicit prompt caching supported.","type":"chatbot","context_length":1048576,"max_completion_tokens":131072,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","max_completion_tokens","presence_penalty","frequency_penalty","seed","n","logprobs","stream","response_format","tools","tool_choice","reasoning_effort","top_k","chat_template_kwargs"],"supported_formats":["openai","anthropic"],"pricing":{"prompt":"6920000000000","completion":"21760000000000"},"pricing_usd":{"prompt":"0.0000014","completion":"0.0000044"},"verifiability":"TeeML","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":3},{"id":"glm-5.3-flash","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"GLM-5.3-Flash","description":"Zhipu AI GLM-5.3-Flash, a fast, cost-effective model for coding and agentic tasks; served via Tencent Cloud MaaS (TokenHub), which exposes both OpenAI and Anthropic faces. Text in / text out, 1M context, up to 128K output. Deep thinking is always on, controllable via reasoning_effort (\"none\" disables it); function calling (tool_choice limited to auto/none in thinking mode), JSON mode (response_format: json_object), and implicit prompt caching supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","max_completion_tokens","presence_penalty","frequency_penalty","seed","n","logprobs","stream","response_format","tools","tool_choice","reasoning_effort"],"supported_formats":["openai","anthropic"],"pricing":{"prompt":"773000000000","completion":"2570000000000"},"pricing_usd":{"prompt":"0.0000001575","completion":"0.000000525"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"glm-5v-turbo","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"GLM-5V-Turbo","description":"Zhipu AI GLM-5V-Turbo, the vision-language turbo variant of GLM-5, served via Tencent Cloud MaaS (TokenHub). Image and text in, text out, 200K-token context, deep-thinking mode, function calling, structured output, and implicit prompt caching.","type":"chatbot","context_length":200000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","presence_penalty","frequency_penalty","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"8830000000000","completion":"29460000000000"},"pricing_usd":{"prompt":"0.0000018","completion":"0.000006"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"hy3","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Hunyuan-3","description":"Tencent Hunyuan 3 (hy3); 295B total / 21B active MoE, native 256K context. Text in / text out. Function calling and implicit prompt caching supported. Served via the Tencent Cloud MaaS (TokenHub international) OpenAI-compatible gateway.","type":"chatbot","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","presence_penalty","frequency_penalty","stop","stream","response_format","tools","tool_choice"],"supported_formats":["openai"],"pricing":{"prompt":"972000000000","completion":"3880000000000"},"pricing_usd":{"prompt":"0.000000198","completion":"0.000000792"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"hy4-preview","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Hunyuan-4-preview","description":"Tencent Hunyuan Hy4 preview; 770B total / 49B active MoE tuned for agent, coding and production workflows — stronger task decomposition, long-horizon tool use and long-chain execution than Hunyuan 3. Served via Tencent Cloud MaaS (TokenHub), which exposes both OpenAI and Anthropic faces. Text in / text out, 1M context, up to 64K output. Deep thinking is on by default and can be disabled with reasoning_effort:\"none\" (enable_thinking is not honored); function calling, JSON mode (response_format: json_object / json_schema) and implicit prompt caching supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","response_format","tools","tool_choice","reasoning_effort"],"supported_formats":["openai","anthropic"],"pricing":{"prompt":"6140000000000","completion":"18420000000000"},"pricing_usd":{"prompt":"0.000001251","completion":"0.0000037515"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"kimi-k2.5","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Kimi-K2.5","description":"Moonshot AI Kimi-K2.5, served through Alibaba Cloud Model Studio (DashScope). Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and implicit prompt caching.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"3270000000000","completion":"17180000000000"},"pricing_usd":{"prompt":"0.000000666667","completion":"0.0000035"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"kimi-k2.6","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Kimi-K2.6","description":"Moonshot AI Kimi-K2.6, served through Alibaba Cloud Model Studio (DashScope). Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, and function calling.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"5310000000000","completion":"22090000000000"},"pricing_usd":{"prompt":"0.000001083333","completion":"0.0000045"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"kimi-k2.7-code","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Kimi-K2.7-Code","description":"Moonshot AI coding model for agentic coding and tool use; multimodal input (text, image, video), thinking always on. 256K context.","type":"chatbot","context_length":262144,"max_completion_tokens":16384,"architecture":{"modality":"text+image+video-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","frequency_penalty","presence_penalty","stop","stream","tools","tool_choice"],"supported_formats":["openai"],"pricing":{"prompt":"5310000000000","completion":"22090000000000"},"pricing_usd":{"prompt":"0.000001083333","completion":"0.0000045"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"kimi-k2.7-code-highspeed","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"Kimi-K2.7-Code-HighSpeed","description":"Moonshot AI Kimi-K2.7-Code-HighSpeed, the high-throughput variant of Kimi-K2.7-Code, served via Tencent Cloud MaaS (TokenHub). Image and text in, text out, 256K-token context, deep thinking, function calling, and implicit prompt caching.","type":"chatbot","context_length":262144,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["max_tokens","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"11890000000000","completion":"50080000000000"},"pricing_usd":{"prompt":"0.0000024225","completion":"0.0000102"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"kimi-k3","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Kimi-K3","description":"Moonshot AI Kimi K3; multimodal input (text, image, video), text output. 1M context, deep thinking always on. Function calling and implicit prompt caching supported.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["top_k","max_tokens","seed","stream","tools","tool_choice","response_format","max_completion_tokens","stop"],"supported_formats":["openai"],"pricing":{"prompt":"14740000000000","completion":"73730000000000"},"pricing_usd":{"prompt":"0.000003","completion":"0.000015"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":3},{"id":"mimo-v2.5-pro","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"MiMo-v2.5-Pro","description":"Xiaomi MiMo-v2.5-Pro, served via Tencent Cloud MaaS (TokenHub). Text in / text out, 1M-token context, thinking always on, function calling, and implicit prompt caching. This endpoint prepends a hidden system prompt of roughly 250 tokens that is billed as input, so even a one-word request reports about 256 prompt tokens.","type":"chatbot","context_length":1048576,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","presence_penalty","frequency_penalty","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"3200000000000","completion":"6400000000000"},"pricing_usd":{"prompt":"0.0000006525","completion":"0.000001305"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"minimax-h3","object":"model","created":1785763398,"owned_by":"0G Foundation","name":"MiniMax-H3","description":"MiniMax-H3 (Hailuo-03) text-to-video and image-to-video, billed per generated second. `seconds` must be 4-15, and outside that range it is silently clamped rather than rejected. `size` names a resolution TIER, not output dimensions: send one of the names listed under Resolution Tiers. OpenAI pixel dimensions (1280x720) are also accepted, but they set only the aspect ratio and only for text-to-video — on image-to-video the ratio follows your reference image, so pixel dimensions have no effect there, while a tier name still selects the tier. Prompt required, up to 7000 characters. Image-to-video takes one first-frame image via multipart `input_reference`. Async: POST /v1/videos, poll GET /v1/videos/{id} until completed, then GET /v1/videos/{id}/content for the MP4.","type":"video-generation","architecture":{"modality":"text+image-\u003evideo","input_modalities":["text","image"],"output_modalities":["video"]},"supported_parameters":["prompt","seconds","size","input_reference"],"supported_formats":["openai"],"default_parameters":{"seconds":4,"size":"2K"},"pricing":{"prompt":"0","completion":"962349910000000000","video":"962349910000000000","variants":[{"dimensions":{"resolution":"2K"},"unit":"video_second","unit_price":"962349910000000000"}]},"pricing_usd":{"prompt":"0","completion":"0","video":"0.195","variants":[{"dimensions":{"resolution":"2K"},"unit":"video_second","unit_price":"0.195"}]},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1},{"id":"minimax-m2.5","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"MiniMax-M2.5","description":"MiniMax-M2.5, served through Alibaba Cloud Model Studio (DashScope). Text in / text out, 192K-token context, thinking always on, function calling, and implicit prompt caching.","type":"chatbot","context_length":196601,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"1710000000000","completion":"6870000000000"},"pricing_usd":{"prompt":"0.00000035","completion":"0.0000014"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"minimax-m2.7","object":"model","created":1789151942,"owned_by":"0G Foundation","name":"MiniMax-M2.7","description":"MiniMax-M2.7, served via Tencent Cloud MaaS (TokenHub). Text in / text out, 200K-token context, thinking always on, function calling, and implicit prompt caching. This endpoint returns its reasoning trace inside message.content wrapped in \u003cthink\u003e tags rather than in a separate reasoning_content field.","type":"chatbot","context_length":200000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other"},"supported_parameters":["temperature","top_p","max_tokens","presence_penalty","frequency_penalty","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"1650000000000","completion":"6620000000000"},"pricing_usd":{"prompt":"0.0000003375","completion":"0.00000135"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"minimax-m3","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"MiniMax-M3","description":"MiniMax-M3, a natively-multimodal model on MiniMax Sparse Attention (MSA); agentic coding, native tool use, and long-horizon tasks. 1M context, thinking on by default.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image+video-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"BPE"},"supported_parameters":["temperature","top_p","max_tokens","stream","tools","tool_choice","thinking","reasoning_split","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"1320000000000","completion":"5300000000000"},"pricing_usd":{"prompt":"0.00000027","completion":"0.00000108"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"gpt-5.5","object":"model","created":1789151945,"owned_by":"0G Foundation","name":"GPT-5.5","description":"GPT-5.5, designed for complex professional workloads, with strong reasoning, high reliability, and improved token efficiency on hard tasks. Text and image input, text output, 1M-token context.","type":"chatbot","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT"},"supported_parameters":["max_tokens","max_completion_tokens","temperature","top_p","stream","tools","tool_choice","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"24570000000000","completion":"147470000000000","cache_write":"30712500000000","cache_write_1h":"49140000000000"},"pricing_usd":{"prompt":"0.000005","completion":"0.00003","cache_write":"0.00000625","cache_write_1h":"0.00001"},"provider_count":1},{"id":"gpt-5.6-luna","object":"model","created":1789151945,"owned_by":"0G Foundation","name":"GPT-5.6 Luna","description":"Fast, cost-efficient model in the GPT-5.6 family, optimized for high-volume, cost-sensitive workloads: responsive chat, classification, extraction, lightweight coding, and agentic workflows at lower latency and cost. Text and image input, text output, 1M-token context.","type":"chatbot","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT"},"supported_parameters":["max_tokens","max_completion_tokens","temperature","top_p","stream","tools","tool_choice","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"983000000000","completion":"5890000000000","cache_write":"1228750000000","cache_write_1h":"1966000000000"},"pricing_usd":{"prompt":"0.0000002","completion":"0.0000012","cache_write":"0.00000025","cache_write_1h":"0.0000004"},"provider_count":1},{"id":"gpt-5.6-sol","object":"model","created":1789151945,"owned_by":"0G Foundation","name":"GPT-5.6 Sol","description":"Flagship of the GPT-5.6 series, built for advanced reasoning, complex coding, and agentic workflows: multi-step software engineering, long-horizon problem solving, and autonomous tool use. Text and image input, text output, 1M-token context.","type":"chatbot","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT"},"supported_parameters":["max_tokens","max_completion_tokens","temperature","top_p","stream","tools","tool_choice","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"24570000000000","completion":"147470000000000","cache_write":"30712500000000","cache_write_1h":"49140000000000"},"pricing_usd":{"prompt":"0.000005","completion":"0.00003","cache_write":"0.00000625","cache_write_1h":"0.00001"},"provider_count":1},{"id":"gpt-5.6-terra","object":"model","created":1789151945,"owned_by":"0G Foundation","name":"GPT-5.6 Terra","description":"Balanced model in the GPT-5.6 family, tuned for workloads that need strong reasoning, coding, and agentic capability at lower cost than the flagship tier. Text and image input, text output, 1M-token context.","type":"chatbot","context_length":1000000,"max_completion_tokens":128000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT"},"supported_parameters":["max_tokens","max_completion_tokens","temperature","top_p","stream","tools","tool_choice","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"9830000000000","completion":"58990000000000","cache_write":"12287500000000","cache_write_1h":"19660000000000"},"pricing_usd":{"prompt":"0.000002","completion":"0.000012","cache_write":"0.0000025","cache_write_1h":"0.000004"},"provider_count":1},{"id":"gpt-6-astra","object":"model","created":1789151945,"owned_by":"0G Foundation","name":"GPT-6 Astra","description":"OpenAI GPT-6 Astra, an agentic model built for long end-to-end tasks: reasoning, software engineering, computer use, browsing, scientific analysis, research, and professional document creation in one model. Text and image input, text output, 1.05M-token context. Sampling controls are fixed upstream, with reasoning_effort as the only tuning knob.","type":"chatbot","context_length":1050000,"max_completion_tokens":128000,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT"},"supported_parameters":["max_completion_tokens","stream","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"49150000000000","completion":"245790000000000","cache_write":"61437500000000","cache_write_1h":"98300000000000"},"pricing_usd":{"prompt":"0.00001","completion":"0.00005","cache_write":"0.0000125","cache_write_1h":"0.00002"},"provider_count":1},{"id":"whisper-large-v3","object":"model","created":1775815334,"owned_by":"0G Foundation","name":"Whisper Large v3","description":"Multilingual automatic speech recognition (ASR); transcription and translation.","type":"speech-to-text","context_length":448,"max_completion_tokens":448,"architecture":{"modality":"audio-\u003etext","input_modalities":["audio"],"output_modalities":["text"]},"supported_parameters":["file","model","language","response_format"],"supported_formats":["openai"],"default_parameters":{"language":"en","response_format":"json"},"pricing":{"prompt":"644100000000000","completion":"0"},"pricing_usd":{"prompt":"0.00013","completion":"0"},"verifiability":"TeeML","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1},{"id":"qwen-flash","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen-Flash","description":"Alibaba Qwen-Flash, the cheapest 1M-token-context model of the Qwen line. Text in / text out, hybrid thinking/non-thinking modes, function calling, built-in tools, and structured output.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"107000000000","completion":"1070000000000"},"pricing_usd":{"prompt":"0.000000021875","completion":"0.00000021875"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen-flash-character","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen-Flash-Character","description":"Alibaba Qwen-Flash-Character, a role-play / character-dialogue model. Text in / text out, 8K-token context. No function calling and no thinking mode.","type":"chatbot","context_length":8192,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"179000000000","completion":"1070000000000"},"pricing_usd":{"prompt":"0.000000036458","completion":"0.00000021875"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen-max","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen-Max","description":"Alibaba Qwen-Max, the top capability tier of the original Qwen line. Text in / text out, 32K-token context, non-thinking only, function calling, built-in tools, and structured output.","type":"chatbot","context_length":32768,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"1710000000000","completion":"6870000000000"},"pricing_usd":{"prompt":"0.00000035","completion":"0.0000014"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen-plus","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen-Plus","description":"Alibaba Qwen-Plus, the long-running balanced model of the Qwen line. Text in / text out, 1M-token context, hybrid thinking/non-thinking modes, function calling, built-in tools, and structured output.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"572000000000","completion":"5720000000000"},"pricing_usd":{"prompt":"0.000000116667","completion":"0.000001166667"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3-vl-30b","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3-VL-30B-A3B-Instruct","description":"Alibaba multimodal vision-language model; strong at visual reasoning, OCR, and document understanding.","type":"chatbot","context_length":262144,"max_completion_tokens":32768,"architecture":{"modality":"text+image-\u003etext","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"default_parameters":{"temperature":0.7,"top_p":0.8},"pricing":{"prompt":"107000000000","completion":"1070000000000"},"pricing_usd":{"prompt":"0.000000021875","completion":"0.00000021875"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3-coder-flash","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3-Coder-Flash","description":"Alibaba Qwen3-Coder-Flash, a fast, low-cost coding model. Text in / text out, 1M-token context, function calling, and structured output.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"716000000000","completion":"2860000000000"},"pricing_usd":{"prompt":"0.000000145833","completion":"0.000000583333"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3-omni-flash","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3-Omni-Flash","description":"Alibaba Qwen3-Omni-Flash, an any-modality-in / text-out model accepting text, image, audio and video input. 48K-token context, hybrid thinking/non-thinking modes, and function calling.","type":"chatbot","context_length":49152,"architecture":{"modality":"text+image+audio+video-\u003etext","input_modalities":["text","image","audio","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"1280000000000","completion":"9090000000000"},"pricing_usd":{"prompt":"0.0000002625","completion":"0.000001852083"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3-vl-plus","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3-VL-Plus","description":"Alibaba Qwen3-VL-Plus, the high-capability tier of the Qwen3-VL vision-language series: image, video and text in, text out. 256K-token context, hybrid thinking/non-thinking modes, function calling, structured output, and implicit prompt caching.","type":"chatbot","context_length":262144,"architecture":{"modality":"text+image+video-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"716000000000","completion":"7160000000000"},"pricing_usd":{"prompt":"0.000000145833","completion":"0.000001458333"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-122b-a10b","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-122B-A10B","description":"Alibaba Qwen3.5-122B-A10B, a 122B-total / 10B-active MoE model of the Qwen3.5 series. Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"572000000000","completion":"4580000000000"},"pricing_usd":{"prompt":"0.000000116667","completion":"0.000000933333"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-27b","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-27B","description":"Alibaba Qwen3.5-27B, the dense 27B model of the Qwen3.5 series. Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"429000000000","completion":"3430000000000"},"pricing_usd":{"prompt":"0.0000000875","completion":"0.0000007"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-35b-a3b","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-35B-A3B","description":"Alibaba Qwen3.5-35B-A3B, a 35B-total / 3B-active MoE model of the Qwen3.5 series. Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"286000000000","completion":"2290000000000"},"pricing_usd":{"prompt":"0.000000058333","completion":"0.000000466667"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-397b-a17b","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-397B-A17B","description":"Alibaba Qwen3.5-397B-A17B, a 397B-total / 17B-active MoE model of the Qwen3.5 series. Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"859000000000","completion":"5150000000000"},"pricing_usd":{"prompt":"0.000000175","completion":"0.00000105"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-flash","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-Flash","description":"Alibaba Qwen3.5-Flash, the fast, low-cost tier of the Qwen3.5 series. Text in / text out, 1M-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"143000000000","completion":"1430000000000"},"pricing_usd":{"prompt":"0.000000029167","completion":"0.000000291667"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-omni-flash","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-Omni-Flash","description":"Alibaba Qwen3.5-Omni-Flash, the fast any-modality-in / text-out model of the Qwen3.5-Omni series, accepting text, image, audio and video input. 192K-token context and function calling.","type":"chatbot","context_length":196608,"architecture":{"modality":"text+image+audio+video-\u003etext","input_modalities":["text","image","audio","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"1570000000000","completion":"9520000000000"},"pricing_usd":{"prompt":"0.000000320833","completion":"0.000001939583"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-omni-plus","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-Omni-Plus","description":"Alibaba Qwen3.5-Omni-Plus, the high-capability any-modality-in / text-out model of the Qwen3.5-Omni series, accepting text, image, audio and video input. 192K-token context and function calling.","type":"chatbot","context_length":196608,"architecture":{"modality":"text+image+audio+video-\u003etext","input_modalities":["text","image","audio","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"5010000000000","completion":"28640000000000"},"pricing_usd":{"prompt":"0.000001020833","completion":"0.000005833333"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.5-plus","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.5-Plus","description":"Alibaba Qwen3.5-Plus, the balanced tier of the Qwen3.5 series. Text in / text out, 1M-token context, hybrid thinking/non-thinking modes, function calling, built-in tools, and structured output.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"572000000000","completion":"3430000000000"},"pricing_usd":{"prompt":"0.000000116667","completion":"0.0000007"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.6-27b","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.6-27B","description":"Alibaba Qwen3.6-27B, the dense 27B model of the Qwen3.6 series. Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"2140000000000","completion":"12890000000000"},"pricing_usd":{"prompt":"0.0000004375","completion":"0.000002625"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.6-35b-a3b","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.6-35B-A3B","description":"Alibaba Qwen3.6-35B-A3B, a 35B-total / 3B-active MoE model of the Qwen3.6 series. Text in / text out, 256K-token context, hybrid thinking/non-thinking modes, function calling, and structured output.","type":"chatbot","context_length":262144,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"1280000000000","completion":"7730000000000"},"pricing_usd":{"prompt":"0.0000002625","completion":"0.000001575"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.6-flash","object":"model","created":1789151941,"owned_by":"0G Foundation","name":"Qwen3.6-Flash","description":"Alibaba Qwen3.6-Flash, a fast, low-cost model of the Qwen3.6 series. Text in / text out, 1M-token context, hybrid thinking/non-thinking modes, function calling, built-in tools, and structured output.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"859000000000","completion":"5150000000000"},"pricing_usd":{"prompt":"0.000000175","completion":"0.00000105"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.6-plus","object":"model","created":1789151939,"owned_by":"0G Foundation","name":"Qwen3.6-Plus","description":"Alibaba Qwen3.6-Plus with hybrid linear attention and sparse MoE; 1M context, 119 languages.","type":"chatbot","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","repetition_penalty","seed","stop","stream","tools","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"default_parameters":{"temperature":0.7,"top_p":0.8},"pricing":{"prompt":"1430000000000","completion":"8590000000000"},"pricing_usd":{"prompt":"0.000000291667","completion":"0.00000175"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.7-flash","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3.7-Flash","description":"Alibaba Qwen3.7-Flash, the fast, low-cost model of the Qwen3.7 series. Text in / text out, 1M-token context, hybrid thinking/non-thinking modes, function calling, built-in tools, structured output, and implicit prompt caching.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"143000000000","completion":"572000000000"},"pricing_usd":{"prompt":"0.000000029167","completion":"0.000000116667"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.7-max","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3.7-Max","description":"Alibaba Qwen3.7-Max with native function calling and web search; 1M context.","type":"chatbot","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","repetition_penalty","seed","stop","stream","tools","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"8590000000000","completion":"25780000000000"},"pricing_usd":{"prompt":"0.00000175","completion":"0.00000525"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.7-plus","object":"model","created":1783298982,"owned_by":"0G Foundation","name":"Qwen3.7-Plus","description":"Alibaba multimodal model with vision and video understanding; native function calling, 1M context.","type":"chatbot","context_length":1000000,"max_completion_tokens":65536,"architecture":{"modality":"text+image+video-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","repetition_penalty","seed","stop","stream","tools","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"1430000000000","completion":"5720000000000"},"pricing_usd":{"prompt":"0.000000291667","completion":"0.000001166667"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.7-text-embedding","object":"model","created":1788872188,"owned_by":"0G Foundation","name":"Qwen3.7 Text Embedding","description":"Alibaba Cloud Bailian's Tongyi Lab text embedding model (Qwen3.7-based), OpenAI Embeddings API compatible (POST /embeddings: model/input/dimensions/encoding_format). Single input up to 128,000 tokens (far above the other Bailian embedding models' 8,192) and user-selectable output dimension. Billed on input tokens only — the response's `usage` carries no completion_tokens, matching this router's `embedding` service type (input-side billing, no output/completion price).","type":"embedding","context_length":128000,"architecture":{"modality":"text-\u003eembedding","input_modalities":["text"],"output_modalities":["embedding"],"tokenizer":"qwen-bpe"},"supported_parameters":["dimensions","encoding_format"],"supported_formats":["openai"],"pricing":{"prompt":"359000000000","completion":"0"},"pricing_usd":{"prompt":"0.0000000735","completion":"0"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1},{"id":"qwen3.8-2.4t-a95b","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3.8-2.4T-A95B","description":"Alibaba Qwen3.8-2.4T-A95B, the 2.4T-parameter / 95B-active MoE member of the Qwen3.8 series. Text in / text out, native 1M-token context, hybrid thinking/non-thinking modes, function calling, structured output, and implicit prompt caching.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format"],"supported_formats":["openai"],"pricing":{"prompt":"8590000000000","completion":"25780000000000"},"pricing_usd":{"prompt":"0.00000175","completion":"0.00000525"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.8-27b","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3.8-27B","description":"Alibaba Qwen3.8-27B, the compact dense model of the Qwen3.8 series. Text in / text out, native 1M-token context, hybrid thinking/non-thinking modes, function calling, structured output, and implicit prompt caching.","type":"chatbot","context_length":1000000,"architecture":{"modality":"text-\u003etext","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"2760000000000","completion":"11040000000000"},"pricing_usd":{"prompt":"0.0000005625","completion":"0.00000225"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.8-flash","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3.8-Flash","description":"Alibaba Qwen3.8-Flash, a fast, cost-effective multimodal model for coding and agentic work; visual and video understanding, function calling, web search, deep thinking, and implicit prompt caching. 1M context.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image+video-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"736000000000","completion":"2480000000000"},"pricing_usd":{"prompt":"0.00000015","completion":"0.00000050625"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"qwen3.8-max","object":"model","created":1789151940,"owned_by":"0G Foundation","name":"Qwen3.8-Max","description":"Alibaba Qwen3.8-Max, a 2.4T-parameter MoE model for coding and agentic work; native visual understanding of documents and video, function calling and web search, 1M context.","type":"chatbot","context_length":1000000,"max_completion_tokens":131072,"architecture":{"modality":"text+image+video-\u003etext","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"qwen-bpe"},"supported_parameters":["temperature","top_p","top_k","max_tokens","presence_penalty","frequency_penalty","repetition_penalty","seed","stop","stream","tools","tool_choice","response_format","enable_thinking","reasoning_effort"],"supported_formats":["openai"],"pricing":{"prompt":"11040000000000","completion":"33140000000000"},"pricing_usd":{"prompt":"0.00000225","completion":"0.00000675"},"verifiability":"TeeTLS","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":2},{"id":"z-image-turbo","object":"model","created":1780905644,"owned_by":"0G Foundation","name":"Z-Image-Turbo","description":"Asynchronous text-to-image model with Base64 output. Generates at most 2 images per request — requesting more (n \u003e 2) returns 2 images, not an error.","type":"text-to-image","context_length":2048,"architecture":{"modality":"text-\u003eimage","input_modalities":["text"],"output_modalities":["image"]},"supported_parameters":["prompt","n","size","response_format"],"supported_formats":["openai"],"default_parameters":{"n":1,"size":"1024x1024"},"pricing":{"prompt":"0","completion":"0","image":"43537840000000000"},"pricing_usd":{"prompt":"0","completion":"0","image":"0.0088"},"verifiability":"TeeML","tee_attested":true,"tee_type":"TDX","tee_verifier":"dstack","provider_count":1}]}