{"object":"list","data":[{"id":"bytedance-seed/seedream-5-0-flash","object":"model","created":1790889880,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seedream-5-0-flash-20261001","hugging_face_id":null,"name":"ByteDance Seed: Seedream 5.0 Flash","description":"Seedream 5.0 Flash is an image generation and editing model from ByteDance Seed. It is the fast, cost-efficient tier of the Seedream 5.0 family, suited for high-volume production and interactive...","context_length":0,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0","image_token":"0.00000431137724550898","image_output":"0.00000431137724550898"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seedream-5-0-flash"}},{"id":"black-forest-labs/flux-3-image","object":"model","created":1790885807,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux-3-image-20261001","hugging_face_id":null,"name":"Black Forest Labs: FLUX.3 Image","description":"FLUX.3 Image is Black Forest Labs' flagship image generation and editing model. It handles text-to-image and multi-reference editing with up to 10 input images, and renders at fixed resolution tiers...","context_length":46864,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000491017964071857","image_output":"0.00000491017964071857"},"top_provider":{"context_length":46864,"max_completion_tokens":42177,"is_moderated":false},"per_request_limits":null,"supported_parameters":["seed"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux-3-image"}},{"id":"liquid/d1","object":"model","created":1790878711,"owned_by":"liquid","canonical_slug":"liquid/d1-20260930","hugging_face_id":null,"name":"LiquidAI: D1","description":"D1 is Liquid AI's structured decision model, served as a System One endpoint. Send a state along with typed questions, and it returns a choice, a score, or a yes/no...","context_length":65536,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000004","completion":"0","input_cache_read":"0.00000004"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/liquid/d1"}},{"id":"apodex/apodex-1.1-mini:free","object":"model","created":1790875335,"owned_by":"apodex","canonical_slug":"apodex/apodex-1.1-mini-20261001","hugging_face_id":null,"name":"Apodex: Apodex 1.1 Mini (free)","description":"Apodex 1.1 Mini is a reasoning-first model from Apodex, built for complex, long-horizon research and forecasting tasks. It works directly with files, data, code, and tools to produce verifiable results,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/apodex/apodex-1.1-mini:free"}},{"id":"microsoft/mai-voice-2.1-flash","object":"model","created":1790870485,"owned_by":"microsoft","canonical_slug":"microsoft/mai-voice-2.1-flash-20261001","hugging_face_id":null,"name":"Microsoft AI: MAI-Voice-2.1-Flash","description":"MAI-Voice-2.1-Flash is a low-latency text-to-speech model from Microsoft AI, optimized for real-time responsiveness. It produces natural, expressive speech across 23 languages, with human-like intonation, rhythm, and emotional nuance. It is...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["cs-CZ-Grant:MAI-Voice-2.1-Flash","cs-CZ-Harper:MAI-Voice-2.1-Flash","da-DK-Grant:MAI-Voice-2.1-Flash","da-DK-Harper:MAI-Voice-2.1-Flash","de-DE-Grant:MAI-Voice-2.1-Flash","de-DE-Harper:MAI-Voice-2.1-Flash","de-DE-Klaus:MAI-Voice-2.1-Flash","de-DE-Mia:MAI-Voice-2.1-Flash","en-AU-Isla:MAI-Voice-2.1-Flash","en-GB-Emily:MAI-Voice-2.1-Flash","en-GB-Harry:MAI-Voice-2.1-Flash","en-IN-Dhruv:MAI-Voice-2.1-Flash","en-IN-Priya:MAI-Voice-2.1-Flash","en-US-Ethan:MAI-Voice-2.1-Flash","en-US-Grant:MAI-Voice-2.1-Flash","en-US-Harper:MAI-Voice-2.1-Flash","en-US-Iris:MAI-Voice-2.1-Flash","en-US-Jasper:MAI-Voice-2.1-Flash","en-US-Olivia:MAI-Voice-2.1-Flash","en-US-Sage:MAI-Voice-2.1-Flash","es-ES-Marta:MAI-Voice-2.1-Flash","es-MX-Alejo:MAI-Voice-2.1-Flash","es-MX-Grant:MAI-Voice-2.1-Flash","es-MX-Harper:MAI-Voice-2.1-Flash","es-MX-Valeria:MAI-Voice-2.1-Flash","fi-FI-Grant:MAI-Voice-2.1-Flash","fi-FI-Harper:MAI-Voice-2.1-Flash","fr-FR-Grant:MAI-Voice-2.1-Flash","fr-FR-Harper:MAI-Voice-2.1-Flash","fr-FR-Marc:MAI-Voice-2.1-Flash","fr-FR-Soleil:MAI-Voice-2.1-Flash","hi-IN-Arjun:MAI-Voice-2.1-Flash","hi-IN-Dhruv:MAI-Voice-2.1-Flash","hi-IN-Grant:MAI-Voice-2.1-Flash","hi-IN-Harper:MAI-Voice-2.1-Flash","hi-IN-Kavya:MAI-Voice-2.1-Flash","hi-IN-Priya:MAI-Voice-2.1-Flash","hu-HU-Bence:MAI-Voice-2.1-Flash","hu-HU-Grant:MAI-Voice-2.1-Flash","hu-HU-Harper:MAI-Voice-2.1-Flash","hu-HU-Levente:MAI-Voice-2.1-Flash","hu-HU-Lilla:MAI-Voice-2.1-Flash","hu-HU-Reka:MAI-Voice-2.1-Flash","id-ID-Grant:MAI-Voice-2.1-Flash","id-ID-Harper:MAI-Voice-2.1-Flash","it-IT-Grant:MAI-Voice-2.1-Flash","it-IT-Harper:MAI-Voice-2.1-Flash","it-IT-Luca:MAI-Voice-2.1-Flash","it-IT-Rosa:MAI-Voice-2.1-Flash","ko-KR-Grant:MAI-Voice-2.1-Flash","ko-KR-Haena:MAI-Voice-2.1-Flash","ko-KR-Harper:MAI-Voice-2.1-Flash","ko-KR-Junho:MAI-Voice-2.1-Flash","nb-NO-Grant:MAI-Voice-2.1-Flash","nb-NO-Harper:MAI-Voice-2.1-Flash","nl-NL-Grant:MAI-Voice-2.1-Flash","nl-NL-Harper:MAI-Voice-2.1-Flash","nl-NL-Sander:MAI-Voice-2.1-Flash","pl-PL-Grant:MAI-Voice-2.1-Flash","pl-PL-Harper:MAI-Voice-2.1-Flash","pt-BR-Caio:MAI-Voice-2.1-Flash","pt-BR-Grant:MAI-Voice-2.1-Flash","pt-BR-Harper:MAI-Voice-2.1-Flash","pt-BR-Luana:MAI-Voice-2.1-Flash","pt-BR-Pedro:MAI-Voice-2.1-Flash","pt-BR-Rafael:MAI-Voice-2.1-Flash","pt-PT-Grant:MAI-Voice-2.1-Flash","pt-PT-Harper:MAI-Voice-2.1-Flash","pt-PT-Rui:MAI-Voice-2.1-Flash","ro-RO-Andrei:MAI-Voice-2.1-Flash","ro-RO-Elena:MAI-Voice-2.1-Flash","ro-RO-Grant:MAI-Voice-2.1-Flash","ro-RO-Harper:MAI-Voice-2.1-Flash","ro-RO-Ioana:MAI-Voice-2.1-Flash","ro-RO-Radu:MAI-Voice-2.1-Flash","ru-RU-Grant:MAI-Voice-2.1-Flash","ru-RU-Harper:MAI-Voice-2.1-Flash","ru-RU-Lev:MAI-Voice-2.1-Flash","ru-RU-Masha:MAI-Voice-2.1-Flash","sv-SE-Grant:MAI-Voice-2.1-Flash","sv-SE-Harper:MAI-Voice-2.1-Flash","th-TH-Grant:MAI-Voice-2.1-Flash","th-TH-Harper:MAI-Voice-2.1-Flash","th-TH-Krit:MAI-Voice-2.1-Flash","th-TH-Nattapong:MAI-Voice-2.1-Flash","tr-TR-Aydin:MAI-Voice-2.1-Flash","tr-TR-Elif:MAI-Voice-2.1-Flash","tr-TR-Grant:MAI-Voice-2.1-Flash","tr-TR-Harper:MAI-Voice-2.1-Flash","vi-VN-Grant:MAI-Voice-2.1-Flash","vi-VN-Harper:MAI-Voice-2.1-Flash","zh-CN-Bo:MAI-Voice-2.1-Flash","zh-CN-Grant:MAI-Voice-2.1-Flash","zh-CN-Harper:MAI-Voice-2.1-Flash","zh-CN-Lan:MAI-Voice-2.1-Flash","zh-CN-Mei:MAI-Voice-2.1-Flash","zh-CN-Wei:MAI-Voice-2.1-Flash"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-voice-2.1-flash"}},{"id":"microsoft/mai-voice-2.1","object":"model","created":1790870467,"owned_by":"microsoft","canonical_slug":"microsoft/mai-voice-2.1-20261001","hugging_face_id":null,"name":"Microsoft AI: MAI-Voice-2.1","description":"MAI-Voice-2.1 is Microsoft AI's highest-fidelity, most expressive text-to-speech model. It produces natural, studio-grade speech across 23 languages, with detailed prosody, nuanced expressiveness, and speaker consistency over long-form content. It is...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000022","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["cs-CZ-Grant:MAI-Voice-2.1","cs-CZ-Harper:MAI-Voice-2.1","da-DK-Grant:MAI-Voice-2.1","da-DK-Harper:MAI-Voice-2.1","de-DE-Grant:MAI-Voice-2.1","de-DE-Harper:MAI-Voice-2.1","de-DE-Klaus:MAI-Voice-2.1","de-DE-Mia:MAI-Voice-2.1","en-AU-Isla:MAI-Voice-2.1","en-GB-Emily:MAI-Voice-2.1","en-GB-Harry:MAI-Voice-2.1","en-IN-Dhruv:MAI-Voice-2.1","en-IN-Priya:MAI-Voice-2.1","en-US-Ethan:MAI-Voice-2.1","en-US-Grant:MAI-Voice-2.1","en-US-Harper:MAI-Voice-2.1","en-US-Iris:MAI-Voice-2.1","en-US-Jasper:MAI-Voice-2.1","en-US-Olivia:MAI-Voice-2.1","en-US-Sage:MAI-Voice-2.1","es-ES-Marta:MAI-Voice-2.1","es-MX-Alejo:MAI-Voice-2.1","es-MX-Grant:MAI-Voice-2.1","es-MX-Harper:MAI-Voice-2.1","es-MX-Valeria:MAI-Voice-2.1","fi-FI-Grant:MAI-Voice-2.1","fi-FI-Harper:MAI-Voice-2.1","fr-FR-Grant:MAI-Voice-2.1","fr-FR-Harper:MAI-Voice-2.1","fr-FR-Marc:MAI-Voice-2.1","fr-FR-Soleil:MAI-Voice-2.1","hi-IN-Arjun:MAI-Voice-2.1","hi-IN-Dhruv:MAI-Voice-2.1","hi-IN-Grant:MAI-Voice-2.1","hi-IN-Harper:MAI-Voice-2.1","hi-IN-Kavya:MAI-Voice-2.1","hi-IN-Priya:MAI-Voice-2.1","hu-HU-Bence:MAI-Voice-2.1","hu-HU-Grant:MAI-Voice-2.1","hu-HU-Harper:MAI-Voice-2.1","hu-HU-Levente:MAI-Voice-2.1","hu-HU-Lilla:MAI-Voice-2.1","hu-HU-Reka:MAI-Voice-2.1","id-ID-Grant:MAI-Voice-2.1","id-ID-Harper:MAI-Voice-2.1","it-IT-Grant:MAI-Voice-2.1","it-IT-Harper:MAI-Voice-2.1","it-IT-Luca:MAI-Voice-2.1","it-IT-Rosa:MAI-Voice-2.1","ko-KR-Grant:MAI-Voice-2.1","ko-KR-Haena:MAI-Voice-2.1","ko-KR-Harper:MAI-Voice-2.1","ko-KR-Junho:MAI-Voice-2.1","nb-NO-Grant:MAI-Voice-2.1","nb-NO-Harper:MAI-Voice-2.1","nl-NL-Grant:MAI-Voice-2.1","nl-NL-Harper:MAI-Voice-2.1","nl-NL-Sander:MAI-Voice-2.1","pl-PL-Grant:MAI-Voice-2.1","pl-PL-Harper:MAI-Voice-2.1","pt-BR-Caio:MAI-Voice-2.1","pt-BR-Grant:MAI-Voice-2.1","pt-BR-Harper:MAI-Voice-2.1","pt-BR-Luana:MAI-Voice-2.1","pt-BR-Pedro:MAI-Voice-2.1","pt-BR-Rafael:MAI-Voice-2.1","pt-PT-Grant:MAI-Voice-2.1","pt-PT-Harper:MAI-Voice-2.1","pt-PT-Rui:MAI-Voice-2.1","ro-RO-Andrei:MAI-Voice-2.1","ro-RO-Elena:MAI-Voice-2.1","ro-RO-Grant:MAI-Voice-2.1","ro-RO-Harper:MAI-Voice-2.1","ro-RO-Ioana:MAI-Voice-2.1","ro-RO-Radu:MAI-Voice-2.1","ru-RU-Grant:MAI-Voice-2.1","ru-RU-Harper:MAI-Voice-2.1","ru-RU-Lev:MAI-Voice-2.1","ru-RU-Masha:MAI-Voice-2.1","sv-SE-Grant:MAI-Voice-2.1","sv-SE-Harper:MAI-Voice-2.1","th-TH-Grant:MAI-Voice-2.1","th-TH-Harper:MAI-Voice-2.1","th-TH-Krit:MAI-Voice-2.1","th-TH-Nattapong:MAI-Voice-2.1","tr-TR-Aydin:MAI-Voice-2.1","tr-TR-Elif:MAI-Voice-2.1","tr-TR-Grant:MAI-Voice-2.1","tr-TR-Harper:MAI-Voice-2.1","vi-VN-Grant:MAI-Voice-2.1","vi-VN-Harper:MAI-Voice-2.1","zh-CN-Bo:MAI-Voice-2.1","zh-CN-Grant:MAI-Voice-2.1","zh-CN-Harper:MAI-Voice-2.1","zh-CN-Lan:MAI-Voice-2.1","zh-CN-Mei:MAI-Voice-2.1","zh-CN-Wei:MAI-Voice-2.1"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-voice-2.1"}},{"id":"unbiased/pareto-26.10-preview","object":"model","created":1790863623,"owned_by":"unbiased","canonical_slug":"unbiased/pareto-26.10-preview-20260929","hugging_face_id":null,"name":"Pareto 26.10 Preview","description":"Pareto is a multimodal composite model built for research, coding, and agentic workflows, while delivering frontier-level performance across a broad range of general-purpose tasks. This is a preview of the...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000032","input_cache_read":"0.00000003"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/unbiased/pareto-26.10-preview"}},{"id":"heygen/heygen-video-1","object":"model","created":1790798684,"owned_by":"heygen","canonical_slug":"heygen/heygen-video-1-20260930","hugging_face_id":null,"name":"HeyGen: Video 1","description":"HeyGen Video 1 is a general-purpose video generation model from HeyGen. It renders short clips with synthesized audio (dialogue, ambience, and sound effects) in a single call, working from a...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/heygen/heygen-video-1"}},{"id":"togethercomputer/tev1-4b-experimental","object":"model","created":1790795429,"owned_by":"togethercomputer","canonical_slug":"togethercomputer/tev1-4b-experimental-20260923","hugging_face_id":"togethercomputer/Tev1-4B-experimental","name":"Together: Tev1 4B Experimental","description":"Tev1 4B Experimental is an experimental decision model from Together AI, a supervised fine-tune of Qwen3.5-4B trained to choose one option from a structured state, question, and list of 2-24...","context_length":32768,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000042","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":8,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/togethercomputer/tev1-4b-experimental"}},{"id":"inception/mercury-decide:free","object":"model","created":1790789334,"owned_by":"inception","canonical_slug":"inception/mercury-decide-20260930","hugging_face_id":null,"name":"Inception: Mercury Decide (free)","description":"Mercury Decide is Inception's structured decision model, served as a System One endpoint. Send a state along with typed questions, and it returns a choice, a score, or a yes/no...","context_length":32768,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inception/mercury-decide:free"}},{"id":"voyageai/rerank-3-lite","object":"model","created":1790785569,"owned_by":"voyageai","canonical_slug":"voyageai/rerank-3-lite-20260930","hugging_face_id":null,"name":"VoyageAI by MongoDB: rerank-3-lite","description":"rerank-3-lite is a reranker optimized for both latency and quality and a drop-in upgrade to rerank-2.5-lite, improving on it by 0.94% NDCG@10 on average across domain evaluations and by 1.86%...","context_length":32000,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/rerank-3-lite"}},{"id":"voyageai/rerank-3","object":"model","created":1790785501,"owned_by":"voyageai","canonical_slug":"voyageai/rerank-3-20260930","hugging_face_id":null,"name":"VoyageAI by MongoDB: rerank-3","description":"rerank-3 is a reranker optimized for quality and a drop-in upgrade to rerank-2.5, improving on it by 0.80% NDCG@10 on average across domain evaluations and by 3.35% on long-document evaluations,...","context_length":32000,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/rerank-3"}},{"id":"openai/gpt-6.1-sol-pro","object":"model","created":1790702886,"owned_by":"openai","canonical_slug":"openai/gpt-6.1-sol-pro-20260929","hugging_face_id":null,"name":"OpenAI: GPT-6.1 Sol Pro","description":"GPT-6.1 Sol Pro is the same underlying model as GPT-6.1 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000002","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6.1-sol-pro"}},{"id":"openai/gpt-6.1-sol","object":"model","created":1790702882,"owned_by":"openai","canonical_slug":"openai/gpt-6.1-sol-20260929","hugging_face_id":null,"name":"OpenAI: GPT-6.1 Sol","description":"GPT-6.1 Sol is an upgrade to GPT-6 Sol from OpenAI, positioned below the flagship GPT-6 Astra in the GPT-6 series. It is suited for agentic coding, computer use, document-heavy professional...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000002","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6.1-sol"}},{"id":"anthropic/claude-sonnet-5.5","object":"model","created":1790618686,"owned_by":"anthropic","canonical_slug":"anthropic/claude-sonnet-5.5-20260928","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5.5","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-5.5"}},{"id":"anthropic/claude-sonnet-5.5:batch","object":"model","created":1790618686,"owned_by":"anthropic","canonical_slug":"anthropic/claude-sonnet-5.5-20260928","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5.5 (batch)","description":"Claude Sonnet 5.5 is Anthropic's Sonnet-class model for well-scoped everyday work, succeeding Claude Sonnet 5 as a direct upgrade. It is especially strong at building features, fixing bugs, and producing...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-5.5:batch"}},{"id":"upstage/solar-decide","object":"model","created":1790592657,"owned_by":"upstage","canonical_slug":"upstage/solar-decide-20260928","hugging_face_id":null,"name":"Upstage: Solar Decide","description":"Solar Decide is Upstage's structured decision model, served as a System One endpoint on Solar Mini 4. Send a state along with typed questions, and it returns a choice, a...","context_length":524288,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0","input_cache_read":"0.00000005"},"top_provider":{"context_length":524288,"max_completion_tokens":471859,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/upstage/solar-decide"}},{"id":"respan/span-01","object":"model","created":1790387550,"owned_by":"respan","canonical_slug":"respan/span-01-20260925","hugging_face_id":null,"name":"Respan: Span-01","description":"Span-01 is a behavior scoring model from Respan. It reads a conversation span and returns, for each plain-language behavior you define, the probability that the behavior is present. It is...","context_length":0,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000002","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/respan/span-01"}},{"id":"respan/span-01-lite","object":"model","created":1790387542,"owned_by":"respan","canonical_slug":"respan/span-01-lite-20260925","hugging_face_id":null,"name":"Respan: Span-01 Lite","description":"Span-01 Lite is the free, lighter tier of Span-01, a behavior scoring model from Respan. It returns, for each plain-language behavior you define, the probability that the behavior is present...","context_length":0,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/respan/span-01-lite"}},{"id":"respan/span-01-lite:free","object":"model","created":1790387542,"owned_by":"respan","canonical_slug":"respan/span-01-lite-20260925","hugging_face_id":null,"name":"Respan: Span-01 Lite (free)","description":"Span-01 Lite is the free, lighter tier of Span-01, a behavior scoring model from Respan. It returns, for each plain-language behavior you define, the probability that the behavior is present...","context_length":0,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/respan/span-01-lite:free"}},{"id":"bytedance-seed/seed-audio-1-0","object":"model","created":1790372476,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-audio-1-0-20260630","hugging_face_id":null,"name":"ByteDance Seed: Seed Audio 1.0","description":"Seed Audio 1.0 is ByteDance Seed's non-streaming audio generation model. It produces speech and other audio from a natural-language text prompt that can describe the desired voice, tone, and sound...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0.0025"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seed-audio-1-0"}},{"id":"typesafe/jev-router","object":"model","created":1790363560,"owned_by":"typesafe","canonical_slug":"typesafe/jev-router","hugging_face_id":null,"name":"TypeSafe: Jev Router","description":"Jev Router picks the best model and reasoning effort for each request, balancing quality, speed, and cost. It runs on Jev, TypeSafe's first System One model, and adapts as your...","context_length":1000000,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"-1","completion":"-1"},"top_provider":{"context_length":null,"max_completion_tokens":null,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/typesafe/jev-router"}},{"id":"jaredpalmer/kev-4b","object":"model","created":1790354233,"owned_by":"jaredpalmer","canonical_slug":"jaredpalmer/kev-4b-20260924","hugging_face_id":"jaredpalmer/kev-4b","name":"Jared Palmer: Kev 4B","description":"Kev 4B is a small open-weight decision model from Jared Palmer, built as a LoRA adapter and pointer head on Qwen3.5-4B-Base and served over the same /v1/systemone contract as TypeSafe's...","context_length":8192,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000042","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/jaredpalmer/kev-4b"}},{"id":"perceptron/perceptron-mk1.5","object":"model","created":1790352661,"owned_by":"perceptron","canonical_slug":"perceptron/perceptron-mk1.5-20260925","hugging_face_id":null,"name":"Perceptron: Perceptron Mk1.5","description":"Perceptron Mk1.5 is Perceptron's embodied reasoning model for physical agents. It accepts text, image, video, and audio input, and answers with text plus optional structured annotations: points, boxes, polygons, tracks,...","context_length":36864,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"top_provider":{"context_length":36864,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perceptron/perceptron-mk1.5"}},{"id":"google/gemini-3.5-transcribe","object":"model","created":1790295820,"owned_by":"google","canonical_slug":"google/gemini-3.5-transcribe-20260827","hugging_face_id":null,"name":"Google: Gemini 3.5 Transcribe","description":"Gemini 3.5 Transcribe is a speech-to-text model from Google. It is suited for synchronous transcription that needs word-level timestamps or speaker diarization, with support for up to eight speakers. Audio...","context_length":98304,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012"},"top_provider":{"context_length":98304,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.5-transcribe"}},{"id":"fish-audio/transcribe-1-pro","object":"model","created":1790278301,"owned_by":"fish-audio","canonical_slug":"fish-audio/transcribe-1-pro-20260924","hugging_face_id":null,"name":"Fish Audio: Transcribe 1 Pro","description":"Transcribe 1 Pro is a speech-to-text model from Fish Audio tuned for interviews, meetings, and podcasts. It labels speakers with inline `speaker` markers, preserves emotion and vocal-event cues such as...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0001","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fish-audio/transcribe-1-pro"}},{"id":"fireworks/ember-1","object":"model","created":1790208461,"owned_by":"fireworks","canonical_slug":"fireworks/ember-1-20260923","hugging_face_id":null,"name":"Fireworks: Ember-1","description":"Ember-1 is a specialized reasoning model from Fireworks Research, built on Kimi K3. It is designed to make every token go further: it produces shorter reasoning traces, using roughly 40%...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","input_cache_read":"0.0000003"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fireworks/ember-1"}},{"id":"inclusionai/ming-image-0.1-design-layer","object":"model","created":1790201839,"owned_by":"inclusionai","canonical_slug":"inclusionai/ming-image-0.1-design-layer-20260922","hugging_face_id":"inclusionAI/Ming-Image-0.1-Design-Layer","name":"inclusionAI: Ming Image 0.1 Design Layer","description":"Ming Image 0.1 Design Layer is an image-to-image model from inclusionAI that decomposes a flattened design image into separate RGBA layers, such as a background layer and foreground elements, and...","context_length":0,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0","image_output":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inclusionai/ming-image-0.1-design-layer"}},{"id":"google/gemini-3.8-flash-lite-tts","object":"model","created":1790200276,"owned_by":"google","canonical_slug":"google/gemini-3.8-flash-lite-tts-20260922","hugging_face_id":null,"name":"Google: Gemini 3.8 Flash Lite TTS","description":"Gemini 3.8 Flash Lite TTS is a text-to-speech model from Google and the fast, high-throughput member of the 3.8 TTS family alongside Gemini 3.8 Flash TTS. It is suited for...","context_length":32768,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000006"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["Zephyr","Puck","Charon","Kore","Fenrir","Leda","Orus","Aoede","Callirrhoe","Autonoe","Enceladus","Iapetus","Umbriel","Algieba","Despina","Erinome","Algenib","Rasalgethi","Laomedeia","Achernar","Alnilam","Schedar","Gacrux","Pulcherrima","Achird","Zubenelgenubi","Vindemiatrix","Sadachbia","Sadaltager","Sulafat"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.8-flash-lite-tts"}},{"id":"google/gemini-3.8-flash-tts","object":"model","created":1790200266,"owned_by":"google","canonical_slug":"google/gemini-3.8-flash-tts-20260922","hugging_face_id":null,"name":"Google: Gemini 3.8 Flash TTS","description":"Gemini 3.8 Flash TTS is a text-to-speech model from Google and the successor to Gemini 3.1 Flash TTS Preview. It is the creative tier of the 3.8 TTS family, suited...","context_length":32768,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000009"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["Zephyr","Puck","Charon","Kore","Fenrir","Leda","Orus","Aoede","Callirrhoe","Autonoe","Enceladus","Iapetus","Umbriel","Algieba","Despina","Erinome","Algenib","Rasalgethi","Laomedeia","Achernar","Alnilam","Schedar","Gacrux","Pulcherrima","Achird","Zubenelgenubi","Vindemiatrix","Sadachbia","Sadaltager","Sulafat"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.8-flash-tts"}},{"id":"z-ai/glm-5.3-prime","object":"model","created":1790199651,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.3-prime-20260921","hugging_face_id":null,"name":"Z.ai: GLM 5.3 Prime","description":"GLM-5.3-Prime is the high-speed variant of Z.ai's GLM-5.3, inheriting its full capabilities while delivering 1.5–2× the output throughput through inference acceleration. It supports text input and output with a 1M-token...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000028","completion":"0.0000088","input_cache_read":"0.00000056"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.3-prime"}},{"id":"qwen/qwen3.8-max-prime","object":"model","created":1790191228,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-max-prime-20260923","hugging_face_id":null,"name":"Qwen: Qwen3.8 Max Prime","description":"Qwen3.8 Max Prime is a higher-throughput variant of Qwen3.8 Max from Alibaba's Qwen team, served as a separate SKU at a higher price point. It accepts text, image, and video...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000005"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-max-prime"}},{"id":"recraft/recraft-v4.1-flash","object":"model","created":1790177946,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-flash-20260923","hugging_face_id":null,"name":"Recraft: Recraft V4.1 Flash","description":"Recraft V4.1 Flash is a text-to-image model from Recraft, the speed and cost tier of the V4.1 family. It generates ~1K raster images in about 1.5 seconds end to end,...","context_length":65536,"architecture":{"modality":"text->image","input_modalities":["text"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000167664670658683","image_output":"0.00000167664670658683"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1-flash"}},{"id":"stealth/space-bunny-alpha","object":"model","created":1790174884,"owned_by":"stealth","canonical_slug":"stealth/space-bunny-alpha","hugging_face_id":null,"name":"Space Bunny Alpha","description":"Space Bunny Alpha is an anonymous large model with blazing-fast inference, strong coding capabilities and native multimodal input support. It delivers adjustable reasoning effort, and a 1M-token context window. Space...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1000000,"max_completion_tokens":524288,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-05","links":{"details":"/v1/models/stealth/space-bunny-alpha"}},{"id":"aion-labs/aion-3.5-mini","object":"model","created":1790170962,"owned_by":"aion-labs","canonical_slug":"aion-labs/aion-3.5-mini-20260923","hugging_face_id":null,"name":"AionLabs: Aion 3.5 Mini","description":"Aion 3.5 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It is the smaller, lower-cost sibling of Aion 3.5 and uses...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000007","completion":"0.0000014","input_cache_read":"0.00000018"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/aion-labs/aion-3.5-mini"}},{"id":"aion-labs/aion-3.5","object":"model","created":1790170961,"owned_by":"aion-labs","canonical_slug":"aion-labs/aion-3.5-20260923","hugging_face_id":null,"name":"AionLabs: Aion 3.5","description":"Aion 3.5 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000006","input_cache_read":"0.00000075"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/aion-labs/aion-3.5"}},{"id":"upstage/solar-mini4","object":"model","created":1790160358,"owned_by":"upstage","canonical_slug":"upstage/solar-mini4-20260922","hugging_face_id":null,"name":"Upstage: Solar Mini 4","description":"Solar Mini 4 is Upstage's compact, cost-efficient language model, a 35B-parameter mixture-of-experts with 3B active parameters and a 524K context window. It is built for agentic use cases where response...","context_length":524288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.000000005"},"top_provider":{"context_length":524288,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/upstage/solar-mini4"}},{"id":"cohere/command-a-plus","object":"model","created":1790102896,"owned_by":"cohere","canonical_slug":"cohere/command-a-plus-05-2026","hugging_face_id":null,"name":"Cohere: Command A+","description":"Command A+ is Cohere's flagship model for enterprise agentic workflows. It accepts text and image inputs with a 192K context window, supports native tool calling with strict tool schemas, structured...","context_length":192000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000015","input_cache_read":"0.00000015"},"top_provider":{"context_length":192000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/cohere/command-a-plus"}},{"id":"openai/gpt-6-luna-pro","object":"model","created":1790100791,"owned_by":"openai","canonical_slug":"openai/gpt-6-luna-pro-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Luna Pro","description":"GPT-6 Luna Pro is the same underlying model as GPT-6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000005","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.00000075","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-luna-pro"}},{"id":"openai/gpt-6-luna-pro:batch","object":"model","created":1790100791,"owned_by":"openai","canonical_slug":"openai/gpt-6-luna-pro-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Luna Pro (batch)","description":"GPT-6 Luna Pro is the same underlying model as GPT-6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-luna-pro:batch"}},{"id":"openai/gpt-6-luna","object":"model","created":1790100786,"owned_by":"openai","canonical_slug":"openai/gpt-6-luna-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Luna","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000005","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.00000075","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-luna"}},{"id":"openai/gpt-6-luna:batch","object":"model","created":1790100786,"owned_by":"openai","canonical_slug":"openai/gpt-6-luna-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Luna (batch)","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.00000025","web_search":"0.01","input_cache_read":"0.000000005","input_cache_write":"0.0000000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000001","completion":"0.000000375","input_cache_read":"0.00000001","input_cache_write":"0.000000125"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-luna:batch"}},{"id":"openai/gpt-6-sol-pro","object":"model","created":1790100781,"owned_by":"openai","canonical_slug":"openai/gpt-6-sol-pro-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Sol Pro","description":"GPT-6 Sol Pro is the same underlying model as GPT-6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-sol-pro"}},{"id":"openai/gpt-6-sol-pro:batch","object":"model","created":1790100781,"owned_by":"openai","canonical_slug":"openai/gpt-6-sol-pro-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Sol Pro (batch)","description":"GPT-6 Sol Pro is the same underlying model as GPT-6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-sol-pro:batch"}},{"id":"openai/gpt-6-sol","object":"model","created":1790100775,"owned_by":"openai","canonical_slug":"openai/gpt-6-sol-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Sol","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-sol"}},{"id":"openai/gpt-6-sol:batch","object":"model","created":1790100775,"owned_by":"openai","canonical_slug":"openai/gpt-6-sol-20260922","hugging_face_id":null,"name":"OpenAI: GPT-6 Sol (batch)","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-sol:batch"}},{"id":"inclusionai/ming-image-0.1-design","object":"model","created":1790095711,"owned_by":"inclusionai","canonical_slug":"inclusionai/ming-image-0.1-design-20260922","hugging_face_id":"inclusionAI/Ming-Image-0.1-Design","name":"inclusionAI: Ming Image 0.1 Design","description":"Ming Image 0.1 Design is a text-to-image model from inclusionAI aimed at graphic-design output, with an emphasis on legible text rendering inside the generated image. It generates from a prompt...","context_length":0,"architecture":{"modality":"text->image","input_modalities":["text"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0","image_output":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inclusionai/ming-image-0.1-design"}},{"id":"anthropic/claude-opus-5.5","object":"model","created":1790094732,"owned_by":"anthropic","canonical_slug":"anthropic/claude-opus-5.5-20260921","hugging_face_id":null,"name":"Anthropic: Claude Opus 5.5","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-5.5"}},{"id":"anthropic/claude-opus-5.5:batch","object":"model","created":1790094732,"owned_by":"anthropic","canonical_slug":"anthropic/claude-opus-5.5-20260921","hugging_face_id":null,"name":"Anthropic: Claude Opus 5.5 (batch)","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-5.5:batch"}},{"id":"assemblyai/universal-3-5-pro","object":"model","created":1790090022,"owned_by":"assemblyai","canonical_slug":"assemblyai/universal-3-5-pro-20260914","hugging_face_id":null,"name":"AssemblyAI: Universal-3.5 Pro","description":"Universal-3.5 Pro is AssemblyAI's speech-to-text model served through its Sync API, returning a complete transcript with word-level timestamps in a single synchronous response for audio clips up to 120 seconds....","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000125","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/assemblyai/universal-3-5-pro"}},{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","object":"model","created":1790021267,"owned_by":"xiaomi","canonical_slug":"xiaomi/mimo-v2.6-pro-ultraspeed-20260921","hugging_face_id":null,"name":"Xiaomi: MiMo-V2.6-Pro-UltraSpeed","description":"MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x...","context_length":1048576,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000435","completion":"0.0000087","input_cache_read":"0.000000036"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":0},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/xiaomi/mimo-v2.6-pro-ultraspeed"}},{"id":"xiaomi/mimo-v2.6-flash","object":"model","created":1790021264,"owned_by":"xiaomi","canonical_slug":"xiaomi/mimo-v2.6-flash-20260921","hugging_face_id":"XiaomiMiMo/MiMo-V2.6-Flash-RL","name":"Xiaomi: MiMo-V2.6-Flash","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":0},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/xiaomi/mimo-v2.6-flash"}},{"id":"xiaomi/mimo-v2.6-pro","object":"model","created":1790021259,"owned_by":"xiaomi","canonical_slug":"xiaomi/mimo-v2.6-pro-20260921","hugging_face_id":"XiaomiMiMo/MiMo-V2.6-Pro-RL","name":"Xiaomi: MiMo-V2.6-Pro","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","video","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":0},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/xiaomi/mimo-v2.6-pro"}},{"id":"x-ai/grok-4.7","object":"model","created":1790007541,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.7-20260916","hugging_face_id":null,"name":"SpaceXAI: Grok 4.7","description":"Grok 4.7 is SpaceXAI's flagship model for coding, agentic tasks, and knowledge work, succeeding Grok 4.6. It is particularly strong at long-running software engineering tasks, verifying its own work, and...","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.7"}},{"id":"qwen/qwen3.8-omni-flash","object":"model","created":1789959954,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-omni-flash-20260918","hugging_face_id":null,"name":"Qwen: Qwen3.8 Omni Flash","description":"Qwen3.8 Omni Flash is an omni-modal reasoning model from Alibaba, the first Qwen model built around agentic capabilities with native audio-video understanding. It is suited for audio-video analysis and summarization,...","context_length":1000000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","image","audio","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-omni-flash"}},{"id":"prism-ml/ternary-bonsai-2-27b","object":"model","created":1789754046,"owned_by":"prism-ml","canonical_slug":"prism-ml/ternary-bonsai-2-27b-20260918","hugging_face_id":"prism-ml/Ternary-Bonsai-2-27B-gguf","name":"PrismML: Ternary Bonsai 2 27B","description":"Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000005","input_cache_read":"0.0000000375"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/prism-ml/ternary-bonsai-2-27b"}},{"id":"z-ai/glm-5.3-flashx","object":"model","created":1789744020,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.3-flashx-20260918","hugging_face_id":"","name":"Z.ai: GLM 5.3 FlashX","description":"GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000037","completion":"0.00000125","input_cache_read":"0.00000009"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.3-flashx"}},{"id":"~typesafe/jev-latest","object":"model","created":1789689685,"owned_by":"~typesafe","canonical_slug":"~typesafe/jev-latest","hugging_face_id":null,"name":"TypeSafe: Jev Latest","description":"This model always redirects to the latest model in the Jev family.","context_length":32000,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000000042","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~typesafe/jev-latest"}},{"id":"typesafe/jev-1.13","object":"model","created":1789689684,"owned_by":"typesafe","canonical_slug":"typesafe/jev-1.13-20260917","hugging_face_id":null,"name":"TypeSafe: Jev 1.13","description":"Jev is a structured decision model from TypeSafe, and the first of its System One models. System One models make fast, structured decisions for software, returning a typed choice rather...","context_length":32000,"architecture":{"modality":"text->decisions","input_modalities":["text"],"output_modalities":["decisions"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000042","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/typesafe/jev-1.13"}},{"id":"unbiased/pareto","object":"model","created":1789686178,"owned_by":"unbiased","canonical_slug":"unbiased/pareto-20260917","hugging_face_id":null,"name":"Pareto","description":"Pareto is a multimodal composite model built for research, coding, and agentic workflows, while delivering frontier-level performance across a broad range of general-purpose tasks.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000075","input_cache_read":"0.00000025"},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/unbiased/pareto"}},{"id":"~deepseek/deepseek-pro-latest","object":"model","created":1789399174,"owned_by":"~deepseek","canonical_slug":"~deepseek/deepseek-pro-latest","hugging_face_id":null,"name":"DeepSeek: DeepSeek Pro Latest","description":"This model always redirects to the latest model in the DeepSeek Pro family.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.0000001325","completion":"0.0000035","input_cache_read":"0.0000001325"},"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~deepseek/deepseek-pro-latest"}},{"id":"~deepseek/deepseek-flash-latest","object":"model","created":1789399150,"owned_by":"~deepseek","canonical_slug":"~deepseek/deepseek-flash-latest","hugging_face_id":null,"name":"DeepSeek: DeepSeek Flash Latest","description":"This model always redirects to the latest model in the DeepSeek Flash family.","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.0000000189","completion":"0.00000075","input_cache_read":"0.0000000189"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~deepseek/deepseek-flash-latest"}},{"id":"inference-net/schematron-v2-turbo","object":"model","created":1789176949,"owned_by":"inference-net","canonical_slug":"inference-net/schematron-v2-turbo-20260902","hugging_face_id":"inference-net/schematron-v2-granite-4.0-h-micro","name":"Inference.net: Schematron V2 Turbo","description":"Schematron V2 Turbo is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes throughput for high-volume extraction workloads. Extraction instructions must be supplied through a JSON schema in response_format rather...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000015","input_cache_read":"0.00000003"},"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inference-net/schematron-v2-turbo"}},{"id":"inference-net/schematron-v2-small","object":"model","created":1789176933,"owned_by":"inference-net","canonical_slug":"inference-net/schematron-v2-small-20260902","hugging_face_id":"inference-net/schematron-v2-llama-3.2-3b","name":"Inference.net: Schematron V2 Small","description":"Schematron V2 Small is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes extraction quality for complex schemas and long pages. Extraction instructions must be supplied through a JSON schema...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.00000023","input_cache_read":"0.00000005"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inference-net/schematron-v2-small"}},{"id":"meta/muse-voice-transcribe-1.0","object":"model","created":1789146931,"owned_by":"meta","canonical_slug":"meta/muse-voice-transcribe-1.0-20260903","hugging_face_id":null,"name":"Meta: Muse Voice Transcribe 1.0","description":"Muse Voice Transcribe 1.0 is a synchronous speech-to-text model from Meta. It is suited for push-to-talk, endpointing, and speaker-aware transcription, with keyword biasing for domain terms and language biasing through...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00005","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-voice-transcribe-1.0"}},{"id":"~openai/gpt-astra-latest","object":"model","created":1789130932,"owned_by":"~openai","canonical_slug":"~openai/gpt-astra-latest","hugging_face_id":null,"name":"OpenAI: GPT Astra Latest","description":"This model always redirects to the latest model in the GPT Astra family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00002","completion":"0.000075","input_cache_read":"0.000002","input_cache_write":"0.000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~openai/gpt-astra-latest"}},{"id":"~openai/gpt-sol-latest","object":"model","created":1789130928,"owned_by":"~openai","canonical_slug":"~openai/gpt-sol-latest","hugging_face_id":null,"name":"OpenAI: GPT Sol Latest","description":"This model always redirects to the latest model in the GPT Sol family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000002","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~openai/gpt-sol-latest"}},{"id":"~openai/gpt-terra-latest","object":"model","created":1789130925,"owned_by":"~openai","canonical_slug":"~openai/gpt-terra-latest","hugging_face_id":null,"name":"OpenAI: GPT Terra Latest","description":"This model always redirects to the latest model in the GPT Terra family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000018","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/~openai/gpt-terra-latest"}},{"id":"~openai/gpt-luna-latest","object":"model","created":1789130922,"owned_by":"~openai","canonical_slug":"~openai/gpt-luna-latest","hugging_face_id":null,"name":"OpenAI: GPT Luna Latest","description":"This model always redirects to the latest model in the GPT Luna family.","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000005","web_search":"0.01","input_cache_read":"0.00000001","input_cache_write":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.00000075","input_cache_read":"0.00000002","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~openai/gpt-luna-latest"}},{"id":"sakana/fugu-ultra-v2","object":"model","created":1789105383,"owned_by":"sakana","canonical_slug":"sakana/fugu-ultra-v2-20260911","hugging_face_id":null,"name":"Sakana: Fugu Ultra v2","description":"Fugu Ultra v2 is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-08-28","expiration_date":null,"links":{"details":"/v1/models/sakana/fugu-ultra-v2"}},{"id":"sakana/fugu-max","object":"model","created":1789104771,"owned_by":"sakana","canonical_slug":"sakana/fugu-max-20260911","hugging_face_id":null,"name":"Sakana: Fugu Max","description":"Fugu Max is the cost-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sakana/fugu-max"}},{"id":"black-forest-labs/flux-video-edit","object":"model","created":1789073532,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux-video-edit-20260910","hugging_face_id":null,"name":"Black Forest Labs: FLUX Video Edit","description":"FLUX Video Edit [fast] takes a source video and an edit prompt and returns a precisely edited video. Add, remove, or replace objects and characters, rebuild the setting, edit on-screen...","context_length":0,"architecture":{"modality":"text+video->video","input_modalities":["text","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux-video-edit"}},{"id":"inclusionai/ling-3.0-flash-vl","object":"model","created":1789056114,"owned_by":"inclusionai","canonical_slug":"inclusionai/ling-3.0-flash-vl-20260910","hugging_face_id":"inclusionAI/Ling-3.0-flash-VL","name":"inclusionAI: Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000021","completion":"0.0000000616","input_cache_read":"0.0000000042"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inclusionai/ling-3.0-flash-vl"}},{"id":"deepseek/deepseek-v4.1-flash","object":"model","created":1789021285,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4.1-flash-20260910","hugging_face_id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek: DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000000189","completion":"0.00000075","input_cache_read":"0.0000000189"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4.1-flash"}},{"id":"deepseek/deepseek-v4.1-flash:batch","object":"model","created":1789021285,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4.1-flash-20260910","hugging_face_id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek: DeepSeek V4.1 Flash (batch)","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000112","completion":"0.000000336","input_cache_read":"0.00000000336"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4.1-flash:batch"}},{"id":"openai/gpt-image-2.5-sunburst","object":"model","created":1788916368,"owned_by":"openai","canonical_slug":"openai/gpt-image-2.5-sunburst-20260908","hugging_face_id":null,"name":"OpenAI: GPT Image 2.5 Sunburst","description":"GPT Image 2.5 Sunburst is an image generation and editing model from OpenAI, positioned as the precision-oriented tier of the GPT Image 2.5 series. It is suited to detailed creative...","context_length":400000,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000008","completion":"0.000008","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"top_provider":{"context_length":400000,"max_completion_tokens":360000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-image-2.5-sunburst"}},{"id":"openai/gpt-image-2.5-flare","object":"model","created":1788916350,"owned_by":"openai","canonical_slug":"openai/gpt-image-2.5-flare-20260908","hugging_face_id":null,"name":"OpenAI: GPT Image 2.5 Flare","description":"GPT Image 2.5 Flare is an image generation and editing model from OpenAI, positioned as the speed-oriented tier of the GPT Image 2.5 series. It is suited to high-volume everyday...","context_length":400000,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000008","completion":"0.000008","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"top_provider":{"context_length":400000,"max_completion_tokens":360000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-image-2.5-flare"}},{"id":"inception/mercury-2.5","object":"model","created":1788892137,"owned_by":"inception","canonical_slug":"inception/mercury-2.5-20260908","hugging_face_id":null,"name":"Inception: Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","context_length":260000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000004","completion":"0.00000015","input_cache_read":"0.000000004"},"top_provider":{"context_length":260000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inception/mercury-2.5"}},{"id":"nex-agi/nex-n2.5-mini","object":"model","created":1788890061,"owned_by":"nex-agi","canonical_slug":"nex-agi/nex-n2.5-mini-20260908","hugging_face_id":"nex-agi/Nex-N2.5-mini","name":"Nex AGI: Nex-N2.5-Mini","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000025","completion":"0.0000001","input_cache_read":"0.0000000025"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.95,"top_k":40},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nex-agi/nex-n2.5-mini"}},{"id":"nex-agi/nex-n2.5-pro","object":"model","created":1788890050,"owned_by":"nex-agi","canonical_slug":"nex-agi/nex-n2.5-pro-20260907","hugging_face_id":"nex-agi/Nex-N2.5-Pro","name":"Nex AGI: Nex-N2.5-Pro","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.00000025","input_cache_read":"0.000000015"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.95,"top_k":40},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nex-agi/nex-n2.5-pro"}},{"id":"openai/gpt-6-astra","object":"model","created":1788552838,"owned_by":"openai","canonical_slug":"openai/gpt-6-astra-20260903","hugging_face_id":null,"name":"OpenAI: GPT-6 Astra","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00002","completion":"0.000075","input_cache_read":"0.000002","input_cache_write":"0.000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-astra"}},{"id":"openai/gpt-6-astra:batch","object":"model","created":1788552838,"owned_by":"openai","canonical_slug":"openai/gpt-6-astra-20260903","hugging_face_id":null,"name":"OpenAI: GPT-6 Astra (batch)","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.0000375","input_cache_read":"0.000001","input_cache_write":"0.0000125"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-astra:batch"}},{"id":"openai/gpt-6-astra-pro","object":"model","created":1788552835,"owned_by":"openai","canonical_slug":"openai/gpt-6-astra-pro-20260903","hugging_face_id":null,"name":"OpenAI: GPT-6 Astra Pro","description":"GPT-6 Astra Pro is the same underlying model as GPT-6 Astra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00002","completion":"0.000075","input_cache_read":"0.000002","input_cache_write":"0.000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-astra-pro"}},{"id":"openai/gpt-6-astra-pro:batch","object":"model","created":1788552835,"owned_by":"openai","canonical_slug":"openai/gpt-6-astra-pro-20260903","hugging_face_id":null,"name":"OpenAI: GPT-6 Astra Pro (batch)","description":"GPT-6 Astra Pro is the same underlying model as GPT-6 Astra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.0000375","input_cache_read":"0.000001","input_cache_write":"0.0000125"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-6-astra-pro:batch"}},{"id":"microsoft/mai-image-2.6","object":"model","created":1788550742,"owned_by":"microsoft","canonical_slug":"microsoft/mai-image-2.6-20260904","hugging_face_id":null,"name":"Microsoft AI: MAI-Image-2.6","description":"MAI-Image-2.6 is an image generation and editing model from Microsoft AI, the precision tier of the MAI-Image-2.6 family alongside the faster [MAI-Image-2.6 Flash](/microsoft/mai-image-2.6-flash). It is suited for design-ready visuals and...","context_length":4096,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0","image_token":"0.000038","image_output":"0.000038"},"top_provider":{"context_length":4096,"max_completion_tokens":1024,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-image-2.6"}},{"id":"microsoft/mai-image-2.6-flash","object":"model","created":1788550735,"owned_by":"microsoft","canonical_slug":"microsoft/mai-image-2.6-flash-20260904","hugging_face_id":null,"name":"Microsoft AI: MAI-Image-2.6 Flash","description":"MAI-Image-2.6 Flash is the lower-latency, lower-cost member of the [MAI-Image-2.6](/microsoft/mai-image-2.6) family from Microsoft AI, built for latency-sensitive, high-throughput production image generation and editing at comparable quality to the precision tier....","context_length":4096,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0","image_token":"0.000019","image_output":"0.000019"},"top_provider":{"context_length":4096,"max_completion_tokens":1024,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-image-2.6-flash"}},{"id":"inclusionai/ling-3.0-flash-sante:free","object":"model","created":1788545946,"owned_by":"inclusionai","canonical_slug":"inclusionai/ling-3.0-flash-sante-20260904","hugging_face_id":null,"name":"inclusionAI: Ling 3.0 Flash Sante (free)","description":"Ling 3.0 Flash Sante is a health and medicine-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inclusionai/ling-3.0-flash-sante:free"}},{"id":"qwen/qwen3.8-max-0902","object":"model","created":1788469704,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-max-20260902","hugging_face_id":null,"name":"Qwen: Qwen3.8 Max (0902)","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025","input_cache_write":"0.0000025"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-max-0902"}},{"id":"microsoft/mai-transcribe-2","object":"model","created":1788396586,"owned_by":"microsoft","canonical_slug":"microsoft/mai-transcribe-2-20260903","hugging_face_id":null,"name":"Microsoft AI: MAI-Transcribe 2","description":"MAI-Transcribe 2 is a multilingual speech-to-text model from Microsoft AI, ranked #1 on the FLEURS multilingual benchmark. It supports 60 languages with automatic language identification, code switching for mixed-language speech,...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.1","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-transcribe-2"}},{"id":"meta/muse-spark-1.3-contributor","object":"model","created":1788381519,"owned_by":"meta","canonical_slug":"meta/muse-spark-1.3-contributor-20260902","hugging_face_id":null,"name":"Meta: Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","context_length":1048576,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000002","web_search":"0.0025","input_cache_read":"0.000000002"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-spark-1.3-contributor"}},{"id":"meta/muse-spark-1.3","object":"model","created":1788378359,"owned_by":"meta","canonical_slug":"meta/muse-spark-1.3-20260902","hugging_face_id":null,"name":"Meta: Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","context_length":1048576,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-spark-1.3"}},{"id":"google/gemini-3.8-flash","object":"model","created":1788362056,"owned_by":"google","canonical_slug":"google/gemini-3.8-flash-20260902","hugging_face_id":null,"name":"Google: Gemini 3.8 Flash","description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.8-flash"}},{"id":"google/gemini-3.8-flash:batch","object":"model","created":1788362056,"owned_by":"google","canonical_slug":"google/gemini-3.8-flash-20260902","hugging_face_id":null,"name":"Google: Gemini 3.8 Flash (batch)","description":"Gemini 3.8 Flash is Google's most intelligent Flash model with significant gains from 3.7 Flash across software engineering, agentic tasks, and multi-step reasoning.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.8-flash:batch"}},{"id":"minimax/hailuo-3-max","object":"model","created":1788313310,"owned_by":"minimax","canonical_slug":"minimax/hailuo-3-max-20260901","hugging_face_id":null,"name":"MiniMax: H3 Max","description":"MiniMax H3 Max is a video-generation model from MiniMax, jointly released with fal.ai. Derived through additional training from MiniMax H3, it is designed for faster text-to-video and image-to-video generation with...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/hailuo-3-max"}},{"id":"anthropic/claude-fable-5.1","object":"model","created":1788285838,"owned_by":"anthropic","canonical_slug":"anthropic/claude-fable-5.1-20260831","hugging_face_id":null,"name":"Anthropic: Claude Fable 5.1","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-fable-5.1"}},{"id":"anthropic/claude-fable-5.1:batch","object":"model","created":1788285838,"owned_by":"anthropic","canonical_slug":"anthropic/claude-fable-5.1-20260831","hugging_face_id":null,"name":"Anthropic: Claude Fable 5.1 (batch)","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.000000125","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-fable-5.1:batch"}},{"id":"ibm-granite/granite-4.2-8b","object":"model","created":1788206780,"owned_by":"ibm-granite","canonical_slug":"ibm-granite/granite-4.2-8b-20260831","hugging_face_id":"ibm-granite/granite-4.2-8b","name":"IBM: Granite 4.2 8B","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000025","input_cache_read":"0.000000015"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/ibm-granite/granite-4.2-8b"}},{"id":"tencent/hy4-preview","object":"model","created":1787897375,"owned_by":"tencent","canonical_slug":"tencent/hy4-preview-20260827","hugging_face_id":"tencent/Hy4-preview","name":"Tencent: Hy4 preview","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000834","completion":"0.000002501","input_cache_read":"0.000000042"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000007506","completion":"0.0000022509","input_cache_read":"0.0000000378"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/tencent/hy4-preview"}},{"id":"alibaba/wan-3.0-prime","object":"model","created":1787863800,"owned_by":"alibaba","canonical_slug":"alibaba/wan-3.0-prime-20260827","hugging_face_id":null,"name":"Alibaba: Wan 3.0 Prime","description":"Wan 3.0 Prime is a fast-mode variant of Wan 3.0 from Alibaba. It supports text-to-video and first-frame image-to-video generation.","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/alibaba/wan-3.0-prime"}},{"id":"inclusionai/ling-3.0-flash-fin","object":"model","created":1787846290,"owned_by":"inclusionai","canonical_slug":"inclusionai/ling-3.0-flash-fin-20260827","hugging_face_id":null,"name":"inclusionAI: Ling 3.0 Flash Fin","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000018","input_cache_read":"0.000000012"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inclusionai/ling-3.0-flash-fin"}},{"id":"~z-ai/glm-flash-latest","object":"model","created":1787817633,"owned_by":"~z-ai","canonical_slug":"~z-ai/glm-flash-latest","hugging_face_id":null,"name":"Z.ai: GLM Flash Latest","description":"This model always redirects to the latest model in the GLM Flash family.","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000002625","completion":"0.000000625","input_cache_read":"0.00000002625"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~z-ai/glm-flash-latest"}},{"id":"qwen/qwen3.8-flash","object":"model","created":1787773060,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-flash-20260826","hugging_face_id":"Qwen/Qwen3.8-Flash-Next","name":"Qwen: Qwen3.8 Flash","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000047","input_cache_read":"0.000000016","input_cache_write":"0.0000002"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-flash"}},{"id":"meta/muse-image","object":"model","created":1787764532,"owned_by":"meta","canonical_slug":"meta/muse-image-1.0-eval-20260824","hugging_face_id":null,"name":"Meta: Muse Image","description":"Muse Image is an agentic image generation model from Meta that generates and edits images from text and reference images. Unlike single-pass image models, it reasons before it renders, breaking...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000239520958083832","image_output":"0.00000239520958083832"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","repetition_penalty","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-image"}},{"id":"z-ai/glm-5.3-flash","object":"model","created":1787752741,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.3-flash-20260826","hugging_face_id":"zai-org/GLM-5.3-Flash","name":"Z.ai: GLM 5.3 Flash","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000005","input_cache_read":"0.00000003"},"top_provider":{"context_length":1048575,"max_completion_tokens":943717,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.3-flash"}},{"id":"z-ai/glm-5.3-flash:batch","object":"model","created":1787752741,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.3-flash-20260826","hugging_face_id":"zai-org/GLM-5.3-Flash","name":"Z.ai: GLM 5.3 Flash (batch)","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.0000002","input_cache_read":"0.000000012"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.3-flash:batch"}},{"id":"recraft/recraft-v4-styles-pro","object":"model","created":1787742640,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-styles-pro-20260826","hugging_face_id":null,"name":"Recraft: Recraft V4 Styles Pro","description":"Recraft V4 Styles Pro is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the reference's...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000239520958083832","image_output":"0.0000239520958083832"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-styles-pro"}},{"id":"recraft/recraft-v4-styles-vector","object":"model","created":1787742635,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-styles-vector-20260826","hugging_face_id":null,"name":"Recraft: Recraft V4 Styles Vector","description":"Recraft V4 Styles Vector is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the reference's...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000119760479041916","image_output":"0.0000119760479041916"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-styles-vector"}},{"id":"recraft/recraft-v4-styles-pro-vector","object":"model","created":1787742628,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-styles-pro-vector-20260826","hugging_face_id":null,"name":"Recraft: Recraft V4 Styles Pro Vector","description":"Recraft V4 Styles Pro Vector is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000287425149700599","image_output":"0.0000287425149700599"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-styles-pro-vector"}},{"id":"recraft/recraft-v4-styles","object":"model","created":1787742458,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-styles-20260826","hugging_face_id":null,"name":"Recraft: Recraft V4 Styles","description":"Recraft V4 Styles is a style-consistent image generation model from Recraft. Every request requires at least one style reference image and generates a new image that reproduces the reference's rendering...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000838323353293413","image_output":"0.00000838323353293413"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-styles"}},{"id":"alibaba/wan-3.0","object":"model","created":1787603856,"owned_by":"alibaba","canonical_slug":"alibaba/wan-3.0-20260824","hugging_face_id":null,"name":"Alibaba: Wan 3.0","description":"Wan 3.0 is a video generation model from Alibaba for text-to-video, image-to-video, and reference-guided video generation. It produces 480p, 720p, or 1080p video with durations from 2 to 30 seconds.","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/alibaba/wan-3.0"}},{"id":"heygen/avatar-iv","object":"model","created":1787595728,"owned_by":"heygen","canonical_slug":"heygen/avatar-iv-20260625","hugging_face_id":null,"name":"HeyGen: Avatar IV","description":"HeyGen: Avatar IV is an image-to-video model that animates a single photo into an expressive, lip-synced talking-head video. Rather than only matching mouth shapes to words, it interprets the vocal...","context_length":0,"architecture":{"modality":"text+image+audio->video","input_modalities":["text","image","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/heygen/avatar-iv"}},{"id":"meta/muse-spark-1.2-contributor","object":"model","created":1787336476,"owned_by":"meta","canonical_slug":"meta/muse-spark-1.2-contributor-20260805","hugging_face_id":null,"name":"Meta: Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","context_length":1048576,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000002","web_search":"0.0025","input_cache_read":"0.000000002"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-spark-1.2-contributor"}},{"id":"deepseek/deepseek-v4-flash-vision-exp","object":"model","created":1787311563,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4-flash-vision-exp-20260821","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek: DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of DeepSeek V4 Flash 0731 from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000002156","completion":"0.0000006468","input_cache_read":"0.00000000686"},"top_provider":{"context_length":1048576,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4-flash-vision-exp"}},{"id":"tencent/hy-mt2-1.8b","object":"model","created":1787231581,"owned_by":"tencent","canonical_slug":"tencent/hy-mt2-1.8b-20260521","hugging_face_id":"tencent/Hy-MT2-1.8B","name":"Tencent: Hy-MT2-1.8B","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000044","completion":"0.000000177"},"top_provider":{"context_length":8192,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","stop","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/tencent/hy-mt2-1.8b"}},{"id":"tencent/hy-mt2-30b-a3b","object":"model","created":1787231561,"owned_by":"tencent","canonical_slug":"tencent/hy-mt2-30b-a3b-20260521","hugging_face_id":"tencent/Hy-MT2-30B-A3B","name":"Tencent: Hy-MT2-30B-A3B","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000074","completion":"0.000000295"},"top_provider":{"context_length":8192,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","stop","structured_outputs","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/tencent/hy-mt2-30b-a3b"}},{"id":"black-forest-labs/flux-video-upscale","object":"model","created":1787173919,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux-video-upscale-20260819","hugging_face_id":null,"name":"Black Forest Labs: FLUX Video Upscale","description":"FLUX Video Upscale is a video upscaling model from Black Forest Labs. It enlarges a single source video by 1.5× to 3× while preserving its duration, with an optional prompt...","context_length":0,"architecture":{"modality":"text+video->video","input_modalities":["text","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux-video-upscale"}},{"id":"~z-ai/glm-latest","object":"model","created":1787151053,"owned_by":"~z-ai","canonical_slug":"~z-ai/glm-latest","hugging_face_id":null,"name":"Z.ai: GLM Latest","description":"This model always redirects to the latest GLM model from Z.ai.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000012","completion":"0.00000114","input_cache_read":"0.000000067"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~z-ai/glm-latest"}},{"id":"tencent/hy-mt2-7b","object":"model","created":1787148797,"owned_by":"tencent","canonical_slug":"tencent/hy-mt2-7b-20260521","hugging_face_id":"tencent/Hy-MT2-7B","name":"Tencent: Hy-MT2-7B","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000074","completion":"0.000000295"},"top_provider":{"context_length":8192,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","stop","structured_outputs","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/tencent/hy-mt2-7b"}},{"id":"z-ai/glm-5.3","object":"model","created":1787086655,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.3-20260816","hugging_face_id":"zai-org/GLM-5.3","name":"Z.ai: GLM 5.3","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000014","completion":"0.0000044","input_cache_read":"0.00000014"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.3"}},{"id":"z-ai/glm-5.3:batch","object":"model","created":1787086655,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.3-20260816","hugging_face_id":"zai-org/GLM-5.3","name":"Z.ai: GLM 5.3 (batch)","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000045","completion":"0.000002","input_cache_read":"0.0000001"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.3:batch"}},{"id":"liquid/lfm-2.5-embedding-350m:free","object":"model","created":1787077908,"owned_by":"liquid","canonical_slug":"liquid/lfm-2.5-embedding-350m-20260818","hugging_face_id":"LiquidAI/LFM2.5-Embedding-350M","name":"LiquidAI: LFM2.5-Embedding-350M (free)","description":"LFM2.5-Embedding-350M is a text embedding model from Liquid AI. It produces 1,024-dimensional embeddings for retrieval and semantic search. Successful CLSSAI requests and embeddings may be retained and used to train...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/liquid/lfm-2.5-embedding-350m:free"}},{"id":"qwen/qwen3.8-27b","object":"model","created":1786722910,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-27b-20260814","hugging_face_id":"Qwen/Qwen3.8-27B","name":"Qwen: Qwen3.8 27B","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000042","completion":"0.000003","input_cache_read":"0.000000085"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-27b"}},{"id":"qwen/qwen3.8-27b:free","object":"model","created":1786722910,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-27b-20260814","hugging_face_id":"Qwen/Qwen3.8-27B","name":"Qwen: Qwen3.8 27B (free)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-27b:free"}},{"id":"dots-studio/dots-3-note-preview:free","object":"model","created":1786680361,"owned_by":"dots-studio","canonical_slug":"dots-studio/dots-3-note-preview-20260813","hugging_face_id":null,"name":"Dots Studio: Dots3-Note Preview (free)","description":"Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...","context_length":512000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":512000,"max_completion_tokens":460800,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-12-31","links":{"details":"/v1/models/dots-studio/dots-3-note-preview:free"}},{"id":"nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b","object":"model","created":1786654371,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b-20260813","hugging_face_id":"nvidia/Nemotron-3.5-ASR-Streaming-Multilingual-0.6b","name":"NVIDIA: Nemotron 3.5 ASR Streaming Multilingual 0.6B","description":"Nemotron 3.5 ASR Streaming Multilingual 0.6B is a speech recognition model from NVIDIA. Its prompt-conditioned, cache-aware FastConformer-RNNT design targets low-latency transcription across more than 40 languages for real-time captioning, voice...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000333","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b"}},{"id":"mistralai/voxtral-small-24b-2507-stt","object":"model","created":1786654002,"owned_by":"mistralai","canonical_slug":"mistralai/voxtral-small-24b-2507-stt-20260813","hugging_face_id":"mistralai/Voxtral-Small-24B-2507","name":"Mistral: Voxtral Small 24B 2507 STT","description":"Voxtral Small 24B 2507 STT is a speech transcription model from Mistral AI. It is suited for transcription, translation, and audio understanding workloads that benefit from its larger model capacity.","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00005","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/voxtral-small-24b-2507-stt"}},{"id":"mistralai/voxtral-mini-3b-2507","object":"model","created":1786653980,"owned_by":"mistralai","canonical_slug":"mistralai/voxtral-mini-3b-2507-20260813","hugging_face_id":"mistralai/Voxtral-Mini-3B-2507","name":"Mistral: Voxtral Mini 3B 2507","description":"Voxtral Mini 3B 2507 is a speech and audio understanding model from Mistral AI. It is suited for transcription, translation, and compact audio processing workloads.","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000166667","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/voxtral-mini-3b-2507"}},{"id":"bytedance-seed/seedream-5-0-lite","object":"model","created":1786650094,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seedream-5-0-lite-20260812","hugging_face_id":null,"name":"ByteDance Seed: Seedream 5.0 Lite","description":"Seedream 5.0 Lite is an image generation model from ByteDance Seed. It is suited for professional visual creation that benefits from web-connected retrieval, complex-prompt comprehension, visual references, and broad knowledge...","context_length":0,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0","image_token":"0.00000838323353293413","image_output":"0.00000838323353293413"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seedream-5-0-lite"}},{"id":"google/gemini-3.7-flash","object":"model","created":1786640581,"owned_by":"google","canonical_slug":"google/gemini-3.7-flash-20260813","hugging_face_id":null,"name":"Google: Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.7-flash"}},{"id":"google/gemini-3.7-flash:batch","object":"model","created":1786640581,"owned_by":"google","canonical_slug":"google/gemini-3.7-flash-20260813","hugging_face_id":null,"name":"Google: Gemini 3.7 Flash (batch)","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.7-flash:batch"}},{"id":"voyageai/voyage-code-4","object":"model","created":1786636912,"owned_by":"voyageai","canonical_slug":"voyageai/voyage-code-4-20260812","hugging_face_id":null,"name":"VoyageAI by MongoDB: voyage-code-4","description":"voyage-code-4 is a code embedding model from Voyage AI, a MongoDB company. It is designed for coding agents and code retrieval, with Matryoshka embeddings at 2048, 1024, 512, and 256...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000012","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/voyage-code-4"}},{"id":"qwen/qwen3-reranker-8b","object":"model","created":1786597684,"owned_by":"qwen","canonical_slug":"qwen/qwen3-reranker-8b","hugging_face_id":"Qwen/Qwen3-Reranker-8B","name":"Qwen3 Reranker 8B","description":"Qwen3 Reranker 8B is a text reranking model from Alibaba Cloud built on the Qwen3 architecture. It evaluates query-document pairs to produce relevance scores for use in retrieval and RAG...","context_length":40960,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":40960,"max_completion_tokens":36864,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-reranker-8b"}},{"id":"qwen/qwen3-asr-1.7b","object":"model","created":1786592646,"owned_by":"qwen","canonical_slug":"qwen/qwen3-asr-1.7b-20260813","hugging_face_id":"Qwen/Qwen3-ASR-1.7B","name":"Qwen: Qwen3 ASR 1.7B","description":"Qwen3 ASR 1.7B is an automatic speech recognition model from Qwen. It supports multilingual language identification and transcription across 30 languages and 22 Chinese dialects, with streaming and offline inference...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000075","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-asr-1.7b"}},{"id":"qwen/qwen3-asr-0.6b","object":"model","created":1786591833,"owned_by":"qwen","canonical_slug":"qwen/qwen3-asr-0.6b-20260813","hugging_face_id":"Qwen/Qwen3-ASR-0.6B","name":"Qwen: Qwen3 ASR 0.6B","description":"Qwen3 ASR 0.6B is a compact automatic speech recognition model from Qwen. It supports multilingual language identification and transcription across 30 languages and 22 Chinese dialects, with streaming and offline...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000333","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-asr-0.6b"}},{"id":"bytedance-seed/seedream-5-0-pro","object":"model","created":1786578139,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seedream-5-0-pro-20260812","hugging_face_id":null,"name":"ByteDance Seed: Seedream 5.0 Pro","description":"Seedream 5.0 Pro is an image generation and editing model from ByteDance Seed. It is suited for commercial visual-production workflows that require precise editing control, lifelike scenes, and natural rendering.","context_length":0,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0.003","image_token":"0.0000107784431137725","image_output":"0.0000107784431137725"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seedream-5-0-pro"}},{"id":"deepgram/flux-tts:free","object":"model","created":1786574888,"owned_by":"deepgram","canonical_slug":"deepgram/flux-tts-20260812","hugging_face_id":null,"name":"Deepgram: Flux TTS (free)","description":"Flux TTS is a text-to-speech model from Deepgram. It is suited for natural, expressive English speech synthesis across Deepgram's Flux voice catalog.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["flux-alexis-en","flux-bree-en","flux-brittany-en","flux-brooke-en","flux-bruce-en","flux-cliff-en","flux-cole-en","flux-colin-en","flux-conor-en","flux-donovan-en","flux-drew-en","flux-elise-en","flux-gemma-en","flux-haley-en","flux-hannah-en","flux-heather-en","flux-jack-en","flux-kai-en","flux-kelsey-en","flux-kit-en","flux-maeve-en","flux-marcelo-en","flux-marcus-en","flux-meena-en","flux-meghan-en","flux-miles-en","flux-naveen-en","flux-paige-en","flux-priya-en","flux-rufus-en","flux-sean-en","flux-sharon-en","flux-sienna-en","flux-tanner-en","flux-wade-en","flux-wes-en"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepgram/flux-tts:free"}},{"id":"bytedance/seedance-2.0-mini","object":"model","created":1786552600,"owned_by":"bytedance","canonical_slug":"bytedance/seedance-2.0-mini-20260811","hugging_face_id":null,"name":"ByteDance: Seedance 2.0 Mini","description":"Seedance 2.0 Mini is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video with image, video, and audio inputs. It...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance/seedance-2.0-mini"}},{"id":"bytedance-seed/seed-2-1-turbo","object":"model","created":1786552176,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-2-1-turbo-20260810","hugging_face_id":null,"name":"ByteDance Seed: Seed 2.1 Turbo","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000025"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seed-2-1-turbo"}},{"id":"qwen/qwen3.8-2.4t-a95b","object":"model","created":1786551702,"owned_by":"qwen","canonical_slug":"qwen/qwen3.8-2.4t-a95b-20260812","hugging_face_id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen: Qwen3.8 2.4T A95B","description":"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.00000025"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.8-2.4t-a95b"}},{"id":"bytedance-seed/seed-2.0-code","object":"model","created":1786550701,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-2.0-code-20260730","hugging_face_id":null,"name":"ByteDance Seed: Seed-2.0-Code","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.000001","completion":"0.000006"}]},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-11-11","links":{"details":"/v1/models/bytedance-seed/seed-2.0-code"}},{"id":"deepseek/deepseek-v4-pro-0813","object":"model","created":1786549364,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4-pro-20260813","hugging_face_id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek: DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044","overrides":[{"utc_days":["saturday","sunday"],"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":0,"utc_end":100,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":100,"utc_end":400,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":400,"utc_end":600,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":600,"utc_end":1000,"prompt":"0.00000132","completion":"0.00000396","input_cache_read":"0.000000044"},{"utc_days":["monday","tuesday","wednesday","thursday","friday"],"utc_start":1000,"utc_end":0,"prompt":"0.00000066","completion":"0.00000198","input_cache_read":"0.000000022"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":393216,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4-pro-0813"}},{"id":"x-ai/grok-4.6","object":"model","created":1786548957,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.6-20260810","hugging_face_id":null,"name":"SpaceXAI: Grok 4.6","description":"Grok 4.6 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM. It is succeeded by [Grok 4.7](/x-ai/grok-4.7).","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.6"}},{"id":"x-ai/grok-imagine-image-2.0","object":"model","created":1786486044,"owned_by":"x-ai","canonical_slug":"x-ai/grok-imagine-image-2.0-20260811","hugging_face_id":null,"name":"xAI: Grok Imagine Image 2.0","description":"Grok Imagine Image 2.0 is an image generation and editing model from xAI. It is suited for creating images from text prompts and editing images from references, with low and...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0.01","image_token":"0.00000958083832335329","image_output":"0.00000958083832335329"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","seed","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-imagine-image-2.0"}},{"id":"liquid/lfm-2.5-2.6b:free","object":"model","created":1786470519,"owned_by":"liquid","canonical_slug":"liquid/lfm-2.5-2.6b-20260811","hugging_face_id":"LiquidAI/LFM2.5-2.6B","name":"LiquidAI: LFM2.5-2.6B (free)","description":"LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":65536,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.1,"top_k":50,"repetition_penalty":1.1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/liquid/lfm-2.5-2.6b:free"}},{"id":"nvidia/nemotron-3.5-lightning","object":"model","created":1786452751,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3.5-lightning-20260807","hugging_face_id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"NVIDIA: Nemotron 3.5 Lightning","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000016","input_cache_read":"0.00000003"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3.5-lightning"}},{"id":"nvidia/nemotron-3.5-lightning:free","object":"model","created":1786452751,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3.5-lightning-20260807","hugging_face_id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"NVIDIA: Nemotron 3.5 Lightning (free)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3.5-lightning:free"}},{"id":"sakana/sakana-namazu","object":"model","created":1786410129,"owned_by":"sakana","canonical_slug":"sakana/namazu-20260811","hugging_face_id":null,"name":"Sakana: Sakana Namazu","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000095","completion":"0.000004","web_search":"0.007","input_cache_read":"0.00000015"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sakana/sakana-namazu"}},{"id":"upstage/solar-pro4","object":"model","created":1786371636,"owned_by":"upstage","canonical_slug":"upstage/solar-pro4-20260810","hugging_face_id":null,"name":"Upstage: Solar Pro 4","description":"Solar Pro 4 is Upstage's cost-efficient large language model, featuring a 524K context window. It is built for long-horizon tasks and agentic workflows, with strong capabilities in office productivity, document-intensive...","context_length":524288,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000036","input_cache_read":"0.000000018"},"top_provider":{"context_length":524288,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/upstage/solar-pro4"}},{"id":"meta/muse-glimmer-30b","object":"model","created":1786302394,"owned_by":"meta","canonical_slug":"meta/muse-glimmer-30b-20260810","hugging_face_id":"meta-models/Muse-Glimmer-30B","name":"Meta: Muse Glimmer 30B","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000035","completion":"0.0000015","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-glimmer-30b"}},{"id":"bytedance/seedance-2.5","object":"model","created":1786141253,"owned_by":"bytedance","canonical_slug":"bytedance/seedance-2.5-20260807","hugging_face_id":null,"name":"ByteDance: Seedance 2.5","description":"Seedance 2.5 is a video generation model from ByteDance. It is suited for long-form storytelling, multimodal reference-based generation, video editing, and video extension. It supports first-frame and first-and-last-frame control, up...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance/seedance-2.5"}},{"id":"openai/gpt-transcribe","object":"model","created":1785973897,"owned_by":"openai","canonical_slug":"openai/gpt-transcribe-20260805","hugging_face_id":null,"name":"OpenAI: GPT Transcribe","description":"GPT Transcribe is a high-accuracy speech-to-text model from OpenAI. It is suited for recorded audio, streamed file transcription, and committed Realtime turns, with free-form context, keyword hints, and multiple language...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000075","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-transcribe"}},{"id":"meta/muse-spark-1.2","object":"model","created":1785959287,"owned_by":"meta","canonical_slug":"meta/muse-spark-1.2-20260805","hugging_face_id":null,"name":"Meta: Muse Spark 1.2","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, and PDF documents, returns text, and offers a 1M-token context window....","context_length":1048576,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-spark-1.2"}},{"id":"qwen/qwen-image-3","object":"model","created":1785894548,"owned_by":"qwen","canonical_slug":"qwen/qwen-image-3-20260805","hugging_face_id":null,"name":"Qwen: Qwen Image 3","description":"Qwen Image 3 is a unified image generation and editing model from Qwen. It supports precise rendering of text and details as small as 10px, along with a richer world...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0.003","image_token":"0.00000718562874251497","image_output":"0.00000718562874251497"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-image-3"}},{"id":"qwen/qwen-image-3-pro","object":"model","created":1785894548,"owned_by":"qwen","canonical_slug":"qwen/qwen-image-3-pro-20260805","hugging_face_id":null,"name":"Qwen: Qwen Image 3 Pro","description":"Qwen Image 3 Pro is an image generation and editing model from Qwen. It supports precise rendering of text and details as small as 10px, along with richer world knowledge...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0.003","image_token":"0.00000958083832335329","image_output":"0.00000958083832335329"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-image-3-pro"}},{"id":"black-forest-labs/flux-3-video","object":"model","created":1785858831,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux-3-video-20260804","hugging_face_id":null,"name":"Black Forest Labs: FLUX.3 Video","description":"FLUX.3 Video is a video generation model from Black Forest Labs. It supports text-to-video, image-guided generation with opening and closing keyframes, and video continuation workflows, making it suited for controlled...","context_length":0,"architecture":{"modality":"text+image+video->video","input_modalities":["text","image","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux-3-video"}},{"id":"~deepseek/deepseek-v4-flash-latest","object":"model","created":1785606009,"owned_by":"~deepseek","canonical_slug":"~deepseek/deepseek-v4-flash-latest","hugging_face_id":null,"name":"DeepSeek: DeepSeek V4 Flash Latest","description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000000465","completion":"0.0000016","input_cache_read":"0.00000000465"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":[],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~deepseek/deepseek-v4-flash-latest"}},{"id":"deepseek/deepseek-v4-flash-0731","object":"model","created":1785478908,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4-flash-20260731","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek: DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows....","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000000062","completion":"0.00000128","input_cache_read":"0.0000000062"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":[],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4-flash-0731"}},{"id":"thinkingmachines/inkling-small","object":"model","created":1785443117,"owned_by":"thinkingmachines","canonical_slug":"thinkingmachines/inkling-small-20260730","hugging_face_id":"thinkingmachines/Inkling-Small","name":"Thinking Machines: Inkling Small","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000045","completion":"0.0000012","input_cache_read":"0.0000001"},"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/thinkingmachines/inkling-small"}},{"id":"thinkingmachines/inkling-small:free","object":"model","created":1785443117,"owned_by":"thinkingmachines","canonical_slug":"thinkingmachines/inkling-small-20260730","hugging_face_id":"thinkingmachines/Inkling-Small","name":"Thinking Machines: Inkling Small (free)","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","context_length":1048576,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","seed","stop","temperature","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/thinkingmachines/inkling-small:free"}},{"id":"minimax/hailuo-3","object":"model","created":1785366648,"owned_by":"minimax","canonical_slug":"minimax/hailuo-03-20260730","hugging_face_id":null,"name":"MiniMax: H3","description":"MiniMax H3 is a lightweight, open-weights video generation model from MiniMax. It is designed for precise multimodal editing and controlled content generation, including instruction-guided edits, text and brand rendering, and...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/hailuo-3"}},{"id":"fish-audio/transcribe-1","object":"model","created":1785353735,"owned_by":"fish-audio","canonical_slug":"fish-audio/transcribe-1-20260729","hugging_face_id":null,"name":"Fish Audio: Transcribe 1","description":"Transcribe 1 is a speech-to-text model from Fish Audio. It is suited for audio transcription with automatic language detection and can return timestamped word-level segments when alignment details are requested.","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0001","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fish-audio/transcribe-1"}},{"id":"fish-audio/s1","object":"model","created":1785353734,"owned_by":"fish-audio","canonical_slug":"fish-audio/s1-20260729","hugging_face_id":null,"name":"Fish Audio: S1","description":"S1 is a multilingual text-to-speech model from Fish Audio. It is suited for voice applications that need broad emotional expression, using parenthetical controls to guide speaking style across its supported...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fish-audio/s1"}},{"id":"fish-audio/s2-pro","object":"model","created":1785353734,"owned_by":"fish-audio","canonical_slug":"fish-audio/s2-pro-20260729","hugging_face_id":null,"name":"Fish Audio: S2 Pro","description":"S2 Pro is a multilingual text-to-speech model from Fish Audio. It is suited for expressive narration and multi-speaker dialogue, with natural-language controls for speaking style and emotion.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fish-audio/s2-pro"}},{"id":"fish-audio/s2.1-pro-free:free","object":"model","created":1785353733,"owned_by":"fish-audio","canonical_slug":"fish-audio/s2.1-pro-free-20260729","hugging_face_id":null,"name":"Fish Audio: S2.1 Pro Free (free)","description":"S2.1 Pro Free is the no-cost variant of Fish Audio S2.1 Pro, intended for testing, prototyping, and low-volume applications. It provides the same synthesis capabilities without production latency or availability...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fish-audio/s2.1-pro-free:free"}},{"id":"fish-audio/s2.1-pro","object":"model","created":1785353732,"owned_by":"fish-audio","canonical_slug":"fish-audio/s2.1-pro-20260729","hugging_face_id":null,"name":"Fish Audio: S2.1 Pro","description":"S2.1 Pro is a production-oriented text-to-speech model from Fish Audio. It is suited for multilingual voice applications, expressive narration, and dialogue synthesis, with open-ended natural-language controls for speaking style and...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/fish-audio/s2.1-pro"}},{"id":"runway/aleph-2","object":"model","created":1785339484,"owned_by":"runway","canonical_slug":"runway/aleph-2-20260729","hugging_face_id":null,"name":"Runway: Aleph 2.0","description":"Runway Aleph 2.0 is an in-context video editing model from Runway. It applies text instructions and keyframe-guided edits across existing footage while preserving details that are not meant to change....","context_length":0,"architecture":{"modality":"text+image+video->video","input_modalities":["text","image","video"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/runway/aleph-2"}},{"id":"runway/gen-4.5","object":"model","created":1785339483,"owned_by":"runway","canonical_slug":"runway/gen-4.5-20260729","hugging_face_id":null,"name":"Runway: Gen-4.5","description":"Runway Gen-4.5 is a video generation model from Runway for text-to-video and image-to-video workflows. It is designed for cinematic scene creation with strong motion quality, visual fidelity, and prompt adherence....","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/runway/gen-4.5"}},{"id":"qwen/qwen3.7-flash","object":"model","created":1785190561,"owned_by":"qwen","canonical_slug":"qwen/qwen3.7-flash-20260727","hugging_face_id":null,"name":"Qwen: Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0.00000013","input_cache_read":"0.000000006","input_cache_write":"0.000000038","overrides":[{"min_prompt_tokens":32000,"prompt":"0.0000001","completion":"0.0000004","input_cache_read":"0.00000002","input_cache_write":"0.000000125"},{"min_prompt_tokens":256000,"prompt":"0.0000002","completion":"0.0000008","input_cache_read":"0.00000004","input_cache_write":"0.00000025"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.7-flash"}},{"id":"voyageai/rerank-2.5-lite","object":"model","created":1785188631,"owned_by":"voyageai","canonical_slug":"voyageai/rerank-2.5-lite-20260727","hugging_face_id":null,"name":"VoyageAI by MongoDB: rerank-2.5-lite","description":"rerank-2.5-lite is a reranker optimized for both latency and quality, delivering a 7.16% improvement in retrieval accuracy over Cohere Rerank v3.5 across 93 datasets. It also outperformed Cohere Rerank v3.5...","context_length":32000,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/rerank-2.5-lite"}},{"id":"voyageai/rerank-2.5","object":"model","created":1785188630,"owned_by":"voyageai","canonical_slug":"voyageai/rerank-2.5-20260727","hugging_face_id":null,"name":"VoyageAI by MongoDB: rerank-2.5","description":"rerank-2.5 is a cutting-edge reranker optimized for quality, delivering a 7.94% improvement in retrieval accuracy over Cohere Rerank v3.5 across 93 datasets. It also outperformed Cohere Rerank v3.5 by 12.70%...","context_length":32000,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/rerank-2.5"}},{"id":"voyageai/voyage-multimodal-3.5","object":"model","created":1785188629,"owned_by":"voyageai","canonical_slug":"voyageai/voyage-multimodal-3.5-20260727","hugging_face_id":null,"name":"VoyageAI by MongoDB: voyage-multimodal-3.5","description":"voyage-multimodal-3.5 is a state-of-the-art multimodal embedding model capable of vectorizing not only text, images, and video individually, but also content that interleaves all three modalities. It delivers excellent performance for...","context_length":32000,"architecture":{"modality":"text+image->embeddings","input_modalities":["text","image"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000012","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/voyage-multimodal-3.5"}},{"id":"voyageai/voyage-4-lite","object":"model","created":1785188627,"owned_by":"voyageai","canonical_slug":"voyageai/voyage-4-lite-20260727","hugging_face_id":null,"name":"VoyageAI by MongoDB: voyage-4-lite","description":"voyage-4-lite is a lightweight, general-purpose embedding model optimized for low latency and cost. Enabled by Matryoshka learning and quantization-aware training, voyage-4-lite supports embeddings in 2048, 1024, 512, and 256 dimensions,...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000002","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/voyage-4-lite"}},{"id":"voyageai/voyage-4","object":"model","created":1785188626,"owned_by":"voyageai","canonical_slug":"voyageai/voyage-4-20260727","hugging_face_id":null,"name":"VoyageAI by MongoDB: voyage-4","description":"voyage-4 is a general-purpose (including multilingual) embedding model optimized for retrieval/search and AI applications. voyage-4 supports embeddings in 2048, 1024, 512, and 256 dimensions, with multiple quantization options. Learn more...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/voyage-4"}},{"id":"voyageai/voyage-4-large","object":"model","created":1785188624,"owned_by":"voyageai","canonical_slug":"voyageai/voyage-4-large-20260727","hugging_face_id":null,"name":"VoyageAI by MongoDB: voyage-4-large","description":"voyage-4-large is a state-of-the-art general-purpose and multilingual embedding optimized for retrieval quality. Enabled by Matryoshka learning and quantization-aware training, voyage-4-large supports embeddings in 2048, 1024, 512, and 256 dimensions, with...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000012","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/voyageai/voyage-4-large"}},{"id":"anthropic/claude-opus-5","object":"model","created":1784912544,"owned_by":"anthropic","canonical_slug":"anthropic/claude-opus-5-20260723","hugging_face_id":null,"name":"Anthropic: Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-5"}},{"id":"anthropic/claude-opus-5:batch","object":"model","created":1784912544,"owned_by":"anthropic","canonical_slug":"anthropic/claude-opus-5-20260723","hugging_face_id":null,"name":"Anthropic: Claude Opus 5 (batch)","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.000003125","input_cache_write_1h":"0.000005"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-5:batch"}},{"id":"microsoft/mai-image-2.5-pro","object":"model","created":1784827701,"owned_by":"microsoft","canonical_slug":"microsoft/mai-image-2.5-pro-20260723","hugging_face_id":null,"name":"Microsoft AI: MAI-Image-2.5 Pro","description":"Microsoft AI's MAI-Image-2.5 is a high-quality image generation model available via Azure AI Foundry. It produces photorealistic and artistic images from text prompts with support for various aspect ratios.","context_length":4096,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0","image_token":"0.000108","image_output":"0.000108"},"top_provider":{"context_length":4096,"max_completion_tokens":1024,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-image-2.5-pro"}},{"id":"microsoft/mai-voice-2-flash","object":"model","created":1784822080,"owned_by":"microsoft","canonical_slug":"microsoft/mai-voice-2-flash-20260723","hugging_face_id":null,"name":"Microsoft AI: MAI-Voice-2-Flash","description":"MAI-Voice-2-Flash is a low-latency text-to-speech model from Microsoft AI for voice agents, assistants, call centers, accessibility, narration, and other interactive applications. It generates expressive 24 kHz mono speech across 15...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["en-US-Harper:MAI-Voice-2","es-MX-Valeria:MAI-Voice-2","fr-FR-Soleil:MAI-Voice-2","de-DE-Klaus:MAI-Voice-2"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-voice-2-flash"}},{"id":"inclusionai/ling-3.0-flash","object":"model","created":1784818580,"owned_by":"inclusionai","canonical_slug":"inclusionai/ling-3.0-flash-20260723","hugging_face_id":"inclusionAI/Ling-3.0-flash","name":"inclusionAI: Ling 3.0 Flash","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000021","completion":"0.000000063","input_cache_read":"0.0000000042"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inclusionai/ling-3.0-flash"}},{"id":"qwen/qwen-audio-3.0-tts-flash","object":"model","created":1784817207,"owned_by":"qwen","canonical_slug":"qwen/qwen-audio-3.0-tts-flash-20260723","hugging_face_id":null,"name":"Qwen: Qwen-Audio-3.0-TTS Flash","description":"Qwen-Audio-3.0-TTS Flash is Alibaba's fast, cost-efficient text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["loongjohn","longanhuan_v3.6"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-audio-3.0-tts-flash"}},{"id":"qwen/qwen-audio-3.0-tts-plus","object":"model","created":1784817207,"owned_by":"qwen","canonical_slug":"qwen/qwen-audio-3.0-tts-plus-20260723","hugging_face_id":null,"name":"Qwen: Qwen-Audio-3.0-TTS Plus","description":"Qwen-Audio-3.0-TTS Plus is Alibaba's higher-quality text-to-speech model, generating spoken audio from text via the DashScope Speech Synthesizer API.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00002","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["longanlingxin","longanlufeng"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-audio-3.0-tts-plus"}},{"id":"x-ai/grok-stt-1.0","object":"model","created":1784817014,"owned_by":"x-ai","canonical_slug":"x-ai/grok-stt-20260723","hugging_face_id":null,"name":"SpaceXAI: Grok STT 1.0","description":"Grok STT is SpaceXAI's speech-to-text model, available via the REST /v1/stt endpoint. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio.","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.0000277777777778","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-stt-1.0"}},{"id":"poolside/laguna-s-2.1","object":"model","created":1784652683,"owned_by":"poolside","canonical_slug":"poolside/laguna-s-2.1-20260720","hugging_face_id":"poolside/Laguna-S-2.1","name":"Poolside: Laguna S 2.1","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000018","input_cache_read":"0.000000009"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/poolside/laguna-s-2.1"}},{"id":"poolside/laguna-s-2.1:free","object":"model","created":1784652683,"owned_by":"poolside","canonical_slug":"poolside/laguna-s-2.1-20260720","hugging_face_id":"poolside/Laguna-S-2.1","name":"Poolside: Laguna S 2.1 (free)","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/poolside/laguna-s-2.1:free"}},{"id":"google/gemini-3.6-flash","object":"model","created":1784646733,"owned_by":"google","canonical_slug":"google/gemini-3.6-flash-20260721","hugging_face_id":null,"name":"Google: Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.6-flash"}},{"id":"google/gemini-3.6-flash:batch","object":"model","created":1784646733,"owned_by":"google","canonical_slug":"google/gemini-3.6-flash-20260721","hugging_face_id":null,"name":"Google: Gemini 3.6 Flash (batch)","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000000375","completion":"0.000001875","image":"0.000000375","audio":"0.000000375","input_audio_cache":"0.0000000375","web_search":"0.014","internal_reasoning":"0.000001875","input_cache_read":"0.0000000375","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.6-flash:batch"}},{"id":"google/gemini-3.5-flash-lite","object":"model","created":1784646726,"owned_by":"google","canonical_slug":"google/gemini-3.5-flash-lite-20260721","hugging_face_id":null,"name":"Google: Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","audio":"0.0000003","input_audio_cache":"0.00000003","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.5-flash-lite"}},{"id":"google/gemini-3.5-flash-lite:batch","object":"model","created":1784646726,"owned_by":"google","canonical_slug":"google/gemini-3.5-flash-lite-20260721","hugging_face_id":null,"name":"Google: Gemini 3.5 Flash Lite (batch)","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000125","image":"0.00000015","audio":"0.00000015","input_audio_cache":"0.000000015","web_search":"0.014","internal_reasoning":"0.00000125","input_cache_read":"0.000000015"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.5-flash-lite:batch"}},{"id":"krea/krea-2-large","object":"model","created":1784574931,"owned_by":"krea","canonical_slug":"krea/krea-2-large-20260720","hugging_face_id":null,"name":"Krea: Krea 2 Large","description":"Krea 2 Large is Krea's high-capability image generation model, more than twice the size of Krea 2 Medium. Its lighter post-training gives images a rawer, more textured, and flexible character,...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000143712574850299","image_output":"0.0000143712574850299"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/krea/krea-2-large"}},{"id":"krea/krea-2-medium","object":"model","created":1784574928,"owned_by":"krea","canonical_slug":"krea/krea-2-medium-20260720","hugging_face_id":null,"name":"Krea: Krea 2 Medium","description":"Krea 2 Medium is Krea's balanced, cost-efficient image generation model and a practical starting point for a broad range of use cases. Its extensive post-training supports stable, consistent generations, with...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000718562874251497","image_output":"0.00000718562874251497"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/krea/krea-2-medium"}},{"id":"krea/krea-2-medium-turbo","object":"model","created":1784574923,"owned_by":"krea","canonical_slug":"krea/krea-2-medium-turbo-20260720","hugging_face_id":null,"name":"Krea: Krea 2 Medium Turbo","description":"Krea 2 Medium Turbo is a distilled, speed-focused variant of Krea 2 Medium from Krea. It is designed for rapid iteration and graphic design exploration where fast generation is the...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Media","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000359281437125749","image_output":"0.00000359281437125749"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/krea/krea-2-medium-turbo"}},{"id":"meituan/longcat-2.0","object":"model","created":1784554658,"owned_by":"meituan","canonical_slug":"meituan/longcat-2.0-20260720","hugging_face_id":"meituan-longcat/LongCat-2.0","name":"Meituan: LongCat 2.0","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","context_length":1048756,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.000000006"},"top_provider":{"context_length":1048756,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meituan/longcat-2.0"}},{"id":"x-ai/grok-imagine-video-1.5","object":"model","created":1784548100,"owned_by":"x-ai","canonical_slug":"x-ai/grok-imagine-video-1.5-20260719","hugging_face_id":null,"name":"SpaceXAI: Grok Imagine Video 1.5","description":"Grok Imagine Video 1.5 is a video generation model from SpaceXAI. It creates videos from text prompts, with an optional starting image to guide the scene. It can direct subject...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","seed","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-imagine-video-1.5"}},{"id":"thinkingmachines/inkling","object":"model","created":1784325956,"owned_by":"thinkingmachines","canonical_slug":"thinkingmachines/inkling-20260715","hugging_face_id":"thinkingmachines/Inkling","name":"Thinking Machines: Inkling","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":524288,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000095","completion":"0.00000405","input_cache_read":"0.00000016"},"top_provider":{"context_length":524288,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/thinkingmachines/inkling"}},{"id":"thinkingmachines/inkling:free","object":"model","created":1784325956,"owned_by":"thinkingmachines","canonical_slug":"thinkingmachines/inkling-20260715","hugging_face_id":"thinkingmachines/Inkling","name":"Thinking Machines: Inkling (free)","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","context_length":1048576,"architecture":{"modality":"text+image+audio->text","input_modalities":["text","image","audio"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":262144,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","seed","stop","temperature","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/thinkingmachines/inkling:free"}},{"id":"deepgram/aura-2","object":"model","created":1784237167,"owned_by":"deepgram","canonical_slug":"deepgram/aura-2-20260716","hugging_face_id":null,"name":"Deepgram: Aura-2","description":"Aura-2 is a multilingual text-to-speech model from Deepgram. It supports Deepgram’s canonical Aura-2 voice catalog for speech synthesis across multiple languages.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":["aura-2-thalia-en","aura-2-agathe-fr","aura-2-agustina-es","aura-2-alvaro-es","aura-2-ama-ja","aura-2-amalthea-en","aura-2-andromeda-en","aura-2-antonia-es","aura-2-apollo-en","aura-2-aquila-es","aura-2-arcas-en","aura-2-aries-en","aura-2-asteria-en","aura-2-athena-en","aura-2-atlas-en","aura-2-aurelia-de","aura-2-aurora-en","aura-2-beatrix-nl","aura-2-callista-en","aura-2-carina-es","aura-2-celeste-es","aura-2-cesare-it","aura-2-cinzia-it","aura-2-cora-en","aura-2-cordelia-en","aura-2-cornelia-nl","aura-2-daphne-nl","aura-2-delia-en","aura-2-demetra-it","aura-2-diana-es","aura-2-dionisio-it","aura-2-draco-en","aura-2-ebisu-ja","aura-2-elara-de","aura-2-electra-en","aura-2-elio-it","aura-2-estrella-es","aura-2-fabian-de","aura-2-flavio-it","aura-2-fujin-ja","aura-2-gloria-es","aura-2-harmonia-en","aura-2-hector-fr","aura-2-helena-en","aura-2-hera-en","aura-2-hermes-en","aura-2-hestia-nl","aura-2-hyperion-en","aura-2-iris-en","aura-2-izanami-ja","aura-2-janus-en","aura-2-javier-es","aura-2-julius-de","aura-2-juno-en","aura-2-jupiter-en","aura-2-kara-de","aura-2-lara-de","aura-2-lars-nl","aura-2-leda-nl","aura-2-livia-it","aura-2-luciano-es","aura-2-luna-en","aura-2-maia-it","aura-2-mars-en","aura-2-melia-it","aura-2-minerva-en","aura-2-neptune-en","aura-2-nestor-es","aura-2-odysseus-en","aura-2-olivia-es","aura-2-ophelia-en","aura-2-orion-en","aura-2-orpheus-en","aura-2-pandora-en","aura-2-phoebe-en","aura-2-pluto-en","aura-2-rhea-nl","aura-2-roman-nl","aura-2-sander-nl","aura-2-saturn-en","aura-2-selena-es","aura-2-selene-en","aura-2-silvia-es","aura-2-sirio-es","aura-2-theia-en","aura-2-uzume-ja","aura-2-valerio-es","aura-2-vesta-en","aura-2-viktoria-de","aura-2-zeus-en"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepgram/aura-2"}},{"id":"moonshotai/kimi-k3","object":"model","created":1784215858,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k3-20260715","hugging_face_id":"moonshotai/Kimi-K3","name":"MoonshotAI: Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001125","completion":"0.0000112","input_cache_read":"0.00000029"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k3"}},{"id":"moonshotai/kimi-k3:batch","object":"model","created":1784215858,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k3-20260715","hugging_face_id":"moonshotai/Kimi-K3","name":"MoonshotAI: Kimi K3 (batch)","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000228","completion":"0.0000114","input_cache_read":"0.000000228"},"top_provider":{"context_length":1048576,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k3:batch"}},{"id":"meta/muse-spark-1.1","object":"model","created":1784215741,"owned_by":"meta","canonical_slug":"meta/muse-spark-1.1-20260709","hugging_face_id":null,"name":"Meta: Muse Spark 1.1","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, and PDF documents and returns text, with a 1M-token context window....","context_length":1048576,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00000425","web_search":"0.0025","input_cache_read":"0.00000015"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/meta/muse-spark-1.1"}},{"id":"nvidia/nemotron-3-embed-1b:free","object":"model","created":1784203294,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-embed-1b-20260716","hugging_face_id":null,"name":"NVIDIA: Nemotron 3 Embed 1B (free)","description":"NVIDIA Nemotron 3 Embed 1B is an open text embedding model from NVIDIA, optimized for high-throughput, low-latency retrieval. It is suited for enterprise search, RAG, code retrieval, and agentic retrieval...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-embed-1b:free"}},{"id":"minimax/speech-2.8-hd","object":"model","created":1784164001,"owned_by":"minimax","canonical_slug":"minimax/speech-2.8-hd-20260716","hugging_face_id":null,"name":"MiniMax: Speech 2.8 HD","description":"MiniMax Speech 2.8 HD is a text-to-speech model from MiniMax. It is suited for applications that generate spoken audio from text and accepts arbitrary MiniMax voice IDs.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0001","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":["English_expressive_narrator","English_radiant_girl","English_magnetic_voiced_man","English_compelling_lady1","English_Aussie_Bloke","English_captivating_female1","English_Upbeat_Woman","English_Trustworth_Man","English_CalmWoman","English_UpsetGirl","English_Gentle-voiced_man","English_Whispering_girl","English_Diligent_Man","English_Graceful_Lady","English_ReservedYoungMan","English_PlayfulGirl","English_ManWithDeepVoice","English_MaturePartner","English_FriendlyPerson","English_MatureBoss","English_Debator","English_LovelyGirl","English_Steadymentor","English_Deep-VoicedGentleman","English_Wiselady","English_CaptivatingStoryteller","English_DecentYoungMan","English_SentimentalLady","English_ImposingManner","English_SadTeen","English_PassionateWarrior","English_WiseScholar","English_Soft-spokenGirl","English_SereneWoman","English_ConfidentWoman","English_PatientMan","English_Comedian","English_BossyLeader","English_Strong-WilledBoy","English_StressedLady","English_AssertiveQueen","English_AnimeCharacter","English_Jovialman","English_WhimsicalGirl","English_Kind-heartedGirl"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/speech-2.8-hd"}},{"id":"minimax/speech-2.8-turbo","object":"model","created":1784164000,"owned_by":"minimax","canonical_slug":"minimax/speech-2.8-turbo-20260716","hugging_face_id":null,"name":"MiniMax: Speech 2.8 Turbo","description":"MiniMax Speech 2.8 Turbo is a text-to-speech model from MiniMax. It is suited for applications that generate spoken audio from text and accepts arbitrary MiniMax voice IDs.","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00006","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":["English_expressive_narrator","English_radiant_girl","English_magnetic_voiced_man","English_compelling_lady1","English_Aussie_Bloke","English_captivating_female1","English_Upbeat_Woman","English_Trustworth_Man","English_CalmWoman","English_UpsetGirl","English_Gentle-voiced_man","English_Whispering_girl","English_Diligent_Man","English_Graceful_Lady","English_ReservedYoungMan","English_PlayfulGirl","English_ManWithDeepVoice","English_MaturePartner","English_FriendlyPerson","English_MatureBoss","English_Debator","English_LovelyGirl","English_Steadymentor","English_Deep-VoicedGentleman","English_Wiselady","English_CaptivatingStoryteller","English_DecentYoungMan","English_SentimentalLady","English_ImposingManner","English_SadTeen","English_PassionateWarrior","English_WiseScholar","English_Soft-spokenGirl","English_SereneWoman","English_ConfidentWoman","English_PatientMan","English_Comedian","English_BossyLeader","English_Strong-WilledBoy","English_StressedLady","English_AssertiveQueen","English_AnimeCharacter","English_Jovialman","English_WhimsicalGirl","English_Kind-heartedGirl"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/speech-2.8-turbo"}},{"id":"deepgram/nova-3","object":"model","created":1784075761,"owned_by":"deepgram","canonical_slug":"deepgram/nova-3-20260714","hugging_face_id":null,"name":"Deepgram: Nova-3","description":"Deepgram Nova-3 general-purpose speech-to-text model with monolingual and multilingual transcription support.","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000716666666667","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepgram/nova-3"}},{"id":"kwaipilot/kat-coder-pro-v2.5","object":"model","created":1783714589,"owned_by":"kwaipilot","canonical_slug":"kwaipilot/kat-coder-pro-v2.5-20260710","hugging_face_id":null,"name":"Kwaipilot: KAT-Coder-Pro V2.5","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000074","completion":"0.00000296","input_cache_read":"0.00000015"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/kwaipilot/kat-coder-pro-v2.5"}},{"id":"openai/gpt-5.6-luna-pro","object":"model","created":1783590867,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Luna Pro","description":"GPT-5.6 Luna Pro is the same underlying model as GPT-5.6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000012","web_search":"0.01","input_cache_read":"0.00000002","input_cache_write":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000004","completion":"0.0000018","input_cache_read":"0.00000004","input_cache_write":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-luna-pro"}},{"id":"openai/gpt-5.6-luna-pro:batch","object":"model","created":1783590867,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-luna-pro-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Luna Pro (batch)","description":"GPT-5.6 Luna Pro is the same underlying model as GPT-5.6 Luna, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-luna-pro:batch"}},{"id":"openai/gpt-5.6-luna","object":"model","created":1783590864,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-luna-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000012","web_search":"0.01","input_cache_read":"0.00000002","input_cache_write":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000004","completion":"0.0000018","input_cache_read":"0.00000004","input_cache_write":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-luna"}},{"id":"openai/gpt-5.6-luna:batch","object":"model","created":1783590864,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-luna-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Luna (batch)","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000006","web_search":"0.01","input_cache_read":"0.00000001","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000002","completion":"0.0000009","input_cache_read":"0.00000002"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-luna:batch"}},{"id":"openai/gpt-5.6-terra-pro","object":"model","created":1783590861,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Terra Pro","description":"GPT-5.6 Terra Pro is the same underlying model as GPT-5.6 Terra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000018","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-terra-pro"}},{"id":"openai/gpt-5.6-terra-pro:batch","object":"model","created":1783590861,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-terra-pro-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Terra Pro (batch)","description":"GPT-5.6 Terra Pro is the same underlying model as GPT-5.6 Terra, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-terra-pro:batch"}},{"id":"openai/gpt-5.6-terra","object":"model","created":1783590857,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-terra-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000018","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-terra"}},{"id":"openai/gpt-5.6-terra:batch","object":"model","created":1783590857,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-terra-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Terra (batch)","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000006","web_search":"0.01","input_cache_read":"0.0000001","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.000009","input_cache_read":"0.0000002"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-terra:batch"}},{"id":"openai/gpt-5.6-sol-pro","object":"model","created":1783590854,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Sol Pro","description":"GPT-5.6 Sol Pro is the same underlying model as GPT-5.6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000004","input_cache_write":"0.000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000008","completion":"0.00003","input_cache_read":"0.0000008","input_cache_write":"0.00001"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-sol-pro"}},{"id":"openai/gpt-5.6-sol-pro:batch","object":"model","created":1783590854,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-sol-pro-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Sol Pro (batch)","description":"GPT-5.6 Sol Pro is the same underlying model as GPT-5.6 Sol, served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. **Cost note:** pro mode spends far more...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-sol-pro:batch"}},{"id":"openai/gpt-5.6-sol","object":"model","created":1783590850,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-sol-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000004","completion":"0.000015","input_cache_read":"0.0000004","input_cache_write":"0.000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-sol"}},{"id":"openai/gpt-5.6-sol:batch","object":"model","created":1783590850,"owned_by":"openai","canonical_slug":"openai/gpt-5.6-sol-20260709","hugging_face_id":null,"name":"OpenAI: GPT-5.6 Sol (batch)","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000002","completion":"0.0000075","input_cache_read":"0.0000002","input_cache_write":"0.0000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2026-02-16","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.6-sol:batch"}},{"id":"x-ai/grok-4.5","object":"model","created":1783523154,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.5-20260708","hugging_face_id":null,"name":"SpaceXAI: Grok 4.5","description":"Grok 4.5 is a model from SpaceXAI with frontier performance on coding, knowledge work, and STEM.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000003","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.0000006"}]},"top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.5"}},{"id":"~x-ai/grok-latest","object":"model","created":1783519360,"owned_by":"~x-ai","canonical_slug":"~x-ai/grok-latest","hugging_face_id":null,"name":"xAI: Grok Latest","description":"This model always redirects to the latest Grok model from xAI.","context_length":500000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","web_search":"0.005","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000012","input_cache_read":"0.000001"}]},"top_provider":{"context_length":500000,"max_completion_tokens":450000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~x-ai/grok-latest"}},{"id":"aion-labs/aion-3.0-mini","object":"model","created":1783443096,"owned_by":"aion-labs","canonical_slug":"aion-labs/aion-3.0-mini-20260707","hugging_face_id":null,"name":"AionLabs: Aion-3.0-Mini","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000007","completion":"0.0000014","input_cache_read":"0.00000018"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/aion-labs/aion-3.0-mini"}},{"id":"aion-labs/aion-3.0","object":"model","created":1783443095,"owned_by":"aion-labs","canonical_slug":"aion-labs/aion-3.0-20260707","hugging_face_id":null,"name":"AionLabs: Aion-3.0","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000006","input_cache_read":"0.00000075"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/aion-labs/aion-3.0"}},{"id":"tencent/hy3","object":"model","created":1783344048,"owned_by":"tencent","canonical_slug":"tencent/hy3-20260706","hugging_face_id":"tencent/Hy3","name":"Tencent: Hy3","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033","overrides":[{"utc_start":0,"utc_end":1600,"prompt":"0.000000132","completion":"0.000000528","input_cache_read":"0.000000033"},{"utc_start":1600,"utc_end":0,"prompt":"0.0000000825","completion":"0.00000033","input_cache_read":"0.000000020625"}]},"top_provider":{"context_length":262144,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.9,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/tencent/hy3"}},{"id":"poolside/laguna-xs-2.1","object":"model","created":1783002429,"owned_by":"poolside","canonical_slug":"poolside/laguna-xs-2.1-20260625","hugging_face_id":"poolside/Laguna-XS-2.1","name":"Poolside: Laguna XS 2.1","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000012","input_cache_read":"0.00000003"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/poolside/laguna-xs-2.1"}},{"id":"poolside/laguna-xs-2.1:free","object":"model","created":1783002429,"owned_by":"poolside","canonical_slug":"poolside/laguna-xs-2.1-20260625","hugging_face_id":"poolside/Laguna-XS-2.1","name":"Poolside: Laguna XS 2.1 (free)","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/poolside/laguna-xs-2.1:free"}},{"id":"anthropic/claude-sonnet-5","object":"model","created":1782843083,"owned_by":"anthropic","canonical_slug":"anthropic/claude-sonnet-5-20260630","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-5"}},{"id":"anthropic/claude-sonnet-5:batch","object":"model","created":1782843083,"owned_by":"anthropic","canonical_slug":"anthropic/claude-sonnet-5-20260630","hugging_face_id":null,"name":"Anthropic: Claude Sonnet 5 (batch)","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-5:batch"}},{"id":"google/gemini-3.1-flash-lite-image","object":"model","created":1782837225,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-lite-image-20260630","hugging_face_id":null,"name":"Google: Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image_output":"0.00003","web_search":"0.014"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-01-01","expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-lite-image"}},{"id":"sakana/fugu-ultra","object":"model","created":1782276303,"owned_by":"sakana","canonical_slug":"sakana/fugu-ultra-20260615","hugging_face_id":null,"name":"Sakana: Fugu Ultra","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","reasoning_effort","structured_outputs","tool_choice","tools","web_search_options"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sakana/fugu-ultra"}},{"id":"alibaba/happyhorse-1.1","object":"model","created":1782269643,"owned_by":"alibaba","canonical_slug":"alibaba/happyhorse-1.1-20260624","hugging_face_id":null,"name":"Alibaba: HappyHorse 1.1","description":"HappyHorse 1.1 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/alibaba/happyhorse-1.1"}},{"id":"openai/gpt-image-2","object":"model","created":1782264714,"owned_by":"openai","canonical_slug":"openai/gpt-image-2","hugging_face_id":null,"name":"OpenAI: GPT Image 2","description":"OpenAI's latest image generation model. Supports high-fidelity image generation and editing via the dedicated Images API.","context_length":400000,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000008","completion":"0.000008","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"top_provider":{"context_length":400000,"max_completion_tokens":360000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-image-2"}},{"id":"openai/gpt-image-1","object":"model","created":1782264713,"owned_by":"openai","canonical_slug":"openai/gpt-image-1","hugging_face_id":null,"name":"OpenAI: GPT Image 1","description":"OpenAI's GPT Image 1 generates and edits images via the dedicated Images API. Features accurate text rendering, transparent backgrounds, and up to 16 reference images for edits.","context_length":400000,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00001","image_output":"0.00004","web_search":"0.01","input_cache_read":"0.00000125"},"top_provider":{"context_length":400000,"max_completion_tokens":360000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-image-1"}},{"id":"openai/gpt-image-1-mini","object":"model","created":1782264713,"owned_by":"openai","canonical_slug":"openai/gpt-image-1-mini","hugging_face_id":null,"name":"OpenAI: GPT Image 1 Mini","description":"A cost-efficient variant of GPT Image 1 for high-quality image generation at reduced latency and cost via OpenAI's dedicated Images API.","context_length":400000,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000025","image_output":"0.000008","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":400000,"max_completion_tokens":360000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-image-1-mini"}},{"id":"alibaba/happyhorse-1.0","object":"model","created":1782260324,"owned_by":"alibaba","canonical_slug":"alibaba/happyhorse-1.0-20260624","hugging_face_id":null,"name":"Alibaba: HappyHorse 1.0","description":"HappyHorse 1.0 is a video generation model from Alibaba. It generates short videos from a text prompt, a single starting image, or a set of reference images, with output up...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/alibaba/happyhorse-1.0"}},{"id":"google/gemini-3.1-flash-image","object":"model","created":1781754065,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-image-20260528","hugging_face_id":null,"name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image)","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","image_output":"0.00006","web_search":"0.014"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-image"}},{"id":"google/gemini-3-pro-image","object":"model","created":1781754054,"owned_by":"google","canonical_slug":"google/gemini-3-pro-image-20260528","hugging_face_id":null,"name":"Google: Nano Banana Pro (Gemini 3 Pro Image)","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":131072,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","image_output":"0.00012","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3-pro-image"}},{"id":"cohere/north-mini-code:free","object":"model","created":1781723748,"owned_by":"cohere","canonical_slug":"cohere/north-mini-code-20260617","hugging_face_id":"CohereLabs/North-Mini-Code-1.0","name":"Cohere: North Mini Code (free)","description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":256000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/cohere/north-mini-code:free"}},{"id":"z-ai/glm-5.2","object":"model","created":1781631930,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.2-20260616","hugging_face_id":"zai-org/GLM-5.2","name":"Z.ai: GLM 5.2","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000041","completion":"0.00000399","input_cache_read":"0.00000026"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.2"}},{"id":"moonshotai/kimi-k2.7-code","object":"model","created":1781266361,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k2.7-code-20260612","hugging_face_id":"moonshotai/Kimi-K2.7-Code","name":"MoonshotAI: Kimi K2.7 Code","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006712","completion":"0.00000335","input_cache_read":"0.00000018"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k2.7-code"}},{"id":"nvidia/llama-nemotron-rerank-vl-1b-v2:free","object":"model","created":1781036054,"owned_by":"nvidia","canonical_slug":"nvidia/llama-nemotron-rerank-vl-1b-v2","hugging_face_id":null,"name":"NVIDIA: Llama Nemotron Rerank VL 1B V2 (free)","description":"Llama Nemotron Rerank VL 1B V2 is a 1.7B multimodal reranking model from NVIDIA. It evaluates the relevance of document images and text against user queries, designed for vision RAG...","context_length":10240,"architecture":{"modality":"text+image->rerank","input_modalities":["text","image"],"output_modalities":["rerank"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":10240,"max_completion_tokens":9216,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/llama-nemotron-rerank-vl-1b-v2:free"}},{"id":"~anthropic/claude-fable-latest","object":"model","created":1781029944,"owned_by":"~anthropic","canonical_slug":"~anthropic/claude-fable-latest","hugging_face_id":null,"name":"Anthropic: Claude Fable Latest","description":"This model always redirects to the latest model in the Claude Fable family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~anthropic/claude-fable-latest"}},{"id":"anthropic/claude-fable-5","object":"model","created":1781007515,"owned_by":"anthropic","canonical_slug":"anthropic/claude-5-fable-20260609","hugging_face_id":null,"name":"Anthropic: Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00005","web_search":"0.01","input_cache_read":"0.000001","input_cache_write":"0.0000125","input_cache_write_1h":"0.00002"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-fable-5"}},{"id":"anthropic/claude-fable-5:batch","object":"model","created":1781007515,"owned_by":"anthropic","canonical_slug":"anthropic/claude-5-fable-20260609","hugging_face_id":null,"name":"Anthropic: Claude Fable 5 (batch)","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-fable-5:batch"}},{"id":"sourceful/riverflow-v2.5-pro","object":"model","created":1780584991,"owned_by":"sourceful","canonical_slug":"sourceful/riverflow-v2.5-pro-20260605","hugging_face_id":null,"name":"Sourceful: Riverflow V2.5 Pro","description":"Riverflow V2.5 Pro is the most powerful variant of Sourceful's Riverflow 2.5 lineup, best for top-tier control and quality-sensitive outputs. The Riverflow 2.5 series is a unified text-to-image and image-to-image...","context_length":32768,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000311377245508982","image_output":"0.0000311377245508982"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","reasoning_effort"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sourceful/riverflow-v2.5-pro"}},{"id":"sourceful/riverflow-v2.5-fast","object":"model","created":1780584983,"owned_by":"sourceful","canonical_slug":"sourceful/riverflow-v2.5-fast-20260605","hugging_face_id":null,"name":"Sourceful: Riverflow V2.5 Fast","description":"Riverflow V2.5 Fast is the speed-optimized variant of Sourceful's Riverflow 2.5 lineup, best for production deployments and latency-critical workflows. The Riverflow 2.5 series is a unified text-to-image and image-to-image family...","context_length":32768,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000455089820359281","image_output":"0.00000455089820359281"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","reasoning","reasoning_effort"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sourceful/riverflow-v2.5-fast"}},{"id":"nvidia/nemotron-3.5-content-safety","object":"model","created":1780581864,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3.5-content-safety-20260604","hugging_face_id":"nvidia/Nemotron-3.5-Content-Safety","name":"NVIDIA: Nemotron 3.5 Content Safety","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3.5-content-safety"}},{"id":"nvidia/nemotron-3.5-content-safety:free","object":"model","created":1780581864,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3.5-content-safety-20260604","hugging_face_id":"nvidia/Nemotron-3.5-Content-Safety","name":"NVIDIA: Nemotron 3.5 Content Safety (free)","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3.5-content-safety:free"}},{"id":"nvidia/nemotron-3-ultra-550b-a55b","object":"model","created":1780551208,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","name":"NVIDIA: Nemotron 3 Ultra","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000022","input_cache_read":"0.0000001"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-ultra-550b-a55b"}},{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","object":"model","created":1780551208,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-ultra-550b-a55b-20260604","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","name":"NVIDIA: Nemotron 3 Ultra (free)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-ultra-550b-a55b:free"}},{"id":"qwen/qwen3.7-plus","object":"model","created":1780491783,"owned_by":"qwen","canonical_slug":"qwen/qwen3.7-plus-20260602","hugging_face_id":null,"name":"Qwen: Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.00000128","input_cache_read":"0.000000064","input_cache_write":"0.0000004","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000096","completion":"0.00000384","input_cache_read":"0.000000192","input_cache_write":"0.0000012"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.7-plus"}},{"id":"microsoft/mai-voice-2","object":"model","created":1780425097,"owned_by":"microsoft","canonical_slug":"microsoft/mai-voice-2","hugging_face_id":null,"name":"Microsoft AI: MAI-Voice-2","description":"MAI-Voice-2 is an expressive text-to-speech model from Microsoft AI. It is suited for conversational assistants, media narration, accessibility, education, and other long-form voice applications. It supports 15 languages across 18...","context_length":0,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000022","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["en-US-Harper:MAI-Voice-2","es-MX-Valeria:MAI-Voice-2","fr-FR-Soleil:MAI-Voice-2","de-DE-Klaus:MAI-Voice-2"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-voice-2"}},{"id":"microsoft/mai-transcribe-1.5","object":"model","created":1780425095,"owned_by":"microsoft","canonical_slug":"microsoft/mai-transcribe-1.5","hugging_face_id":null,"name":"Microsoft AI: MAI-Transcribe 1.5","description":"MAI-Transcribe 1.5 is a multilingual speech-to-text model from Microsoft AI. It is suited for captions, call transcription, subtitling, accessibility, and other voice-enabled applications, with reliable transcription across 43 languages, diverse...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.36","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-transcribe-1.5"}},{"id":"microsoft/mai-image-2.5","object":"model","created":1780424896,"owned_by":"microsoft","canonical_slug":"microsoft/mai-image-2.5","hugging_face_id":null,"name":"Microsoft AI: MAI-Image-2.5","description":"Microsoft AI's MAI-Image-2.5 is a high-quality image generation model available via Azure AI Foundry. It produces photorealistic and artistic images from text prompts with support for various aspect ratios.","context_length":4096,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0","image_token":"0.000047","image_output":"0.000047"},"top_provider":{"context_length":4096,"max_completion_tokens":1024,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","temperature"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/microsoft/mai-image-2.5"}},{"id":"minimax/minimax-m3","object":"model","created":1780245374,"owned_by":"minimax","canonical_slug":"minimax/minimax-m3-20260531","hugging_face_id":"MiniMaxAI/Minimax-M3","name":"MiniMax: MiniMax M3","description":"MiniMax-M3 is a multimodal foundation model from MiniMax. It supports text, image, and video inputs with text output, a 1M-token context window, and is suited for long-horizon agentic work, coding,...","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000006"},"top_provider":{"context_length":524288,"max_completion_tokens":512000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-m3"}},{"id":"stepfun/step-3.7-flash","object":"model","created":1779985069,"owned_by":"stepfun","canonical_slug":"stepfun/step-3.7-flash-20260528","hugging_face_id":"stepfun-ai/Step-3.7-Flash","name":"StepFun: Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.00000115","input_cache_read":"0.00000004"},"top_provider":{"context_length":256000,"max_completion_tokens":230400,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/stepfun/step-3.7-flash"}},{"id":"anthropic/claude-opus-4.8","object":"model","created":1779905091,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.8-opus-20260528","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.8"}},{"id":"anthropic/claude-opus-4.8:batch","object":"model","created":1779905091,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.8-opus-20260528","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.8 (batch)","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.000003125","input_cache_write_1h":"0.000005"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.8:batch"}},{"id":"nvidia/parakeet-tdt-0.6b-v3","object":"model","created":1779848335,"owned_by":"nvidia","canonical_slug":"nvidia/parakeet-tdt-0.6b-v3","hugging_face_id":null,"name":"NVIDIA: Parakeet TDT 0.6B v3","description":"Parakeet TDT 0.6B v3 is NVIDIA's 600M-parameter multilingual speech-to-text model built on the FastConformer-TDT architecture. Trained on the Granary dataset (670,000+ hours of audio), it supports automatic language detection across...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000025","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/parakeet-tdt-0.6b-v3"}},{"id":"qwen/qwen3.7-max","object":"model","created":1779376861,"owned_by":"qwen","canonical_slug":"qwen/qwen3.7-max-20260520","hugging_face_id":null,"name":"Qwen: Qwen3.7 Max","description":"Qwen3.7-Max is the flagship model in Alibaba's Qwen3.7 series. It supports text input and output and is designed for agent-centric workloads, with particular strengths in coding, office and productivity tasks,...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000001475","completion":"0.000004425","input_cache_read":"0.000000295","input_cache_write":"0.00000184375"},"top_provider":{"context_length":1000000,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.7-max"}},{"id":"x-ai/grok-build-0.1","object":"model","created":1779298123,"owned_by":"x-ai","canonical_slug":"x-ai/grok-build-0.1-20260520","hugging_face_id":null,"name":"SpaceXAI: Grok Build 0.1","description":"Grok Build 0.1 is SpaceXAI’s fast coding model trained specifically for agentic software engineering workflows. It supports text and image inputs with text output, and is optimized for interactive coding...","context_length":256000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":256000,"max_completion_tokens":230400,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-build-0.1"}},{"id":"google/gemini-embedding-2","object":"model","created":1779290135,"owned_by":"google","canonical_slug":"google/gemini-embedding-2","hugging_face_id":null,"name":"Google: Gemini Embedding 2","description":"Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0","image":"0.00000045","audio":"0.0000065"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-embedding-2"}},{"id":"google/gemini-embedding-2:batch","object":"model","created":1779290135,"owned_by":"google","canonical_slug":"google/gemini-embedding-2","hugging_face_id":null,"name":"Google: Gemini Embedding 2 (batch)","description":"Gemini Embedding 2 is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It supports...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0","image":"0.000000225","audio":"0.00000325"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-embedding-2:batch"}},{"id":"google/gemini-3.5-flash","object":"model","created":1779193800,"owned_by":"google","canonical_slug":"google/gemini-3.5-flash-20260519","hugging_face_id":null,"name":"Google: Gemini 3.5 Flash","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000015","completion":"0.000009","image":"0.0000015","audio":"0.000003","input_audio_cache":"0.0000003","web_search":"0.014","internal_reasoning":"0.000009","input_cache_read":"0.00000015","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-01","expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.5-flash"}},{"id":"google/gemini-3.5-flash:batch","object":"model","created":1779193800,"owned_by":"google","canonical_slug":"google/gemini-3.5-flash-20260519","hugging_face_id":null,"name":"Google: Gemini 3.5 Flash (batch)","description":"Gemini 3.5 Flash is Google's high-efficiency multimodal model, bringing near-Pro level coding and reasoning at Flash-tier cost and speed. It is highly optimized for coding proficiency and parallel agentic execution...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000045","image":"0.00000075","audio":"0.0000015","input_audio_cache":"0.00000015","web_search":"0.014","internal_reasoning":"0.0000045","input_cache_read":"0.000000075"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-01","expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.5-flash:batch"}},{"id":"x-ai/grok-imagine-video","object":"model","created":1779117586,"owned_by":"x-ai","canonical_slug":"x-ai/grok-imagine-video-20260512","hugging_face_id":null,"name":"SpaceXAI: Grok Imagine Video","description":"Grok Imagine Video is SpaceXAI's fast, text-, image-, and reference-conditioned video generation model. It produces short videos (1–15 seconds, 24 fps) at 480p or 720p across seven aspect ratios -...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","seed","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-imagine-video"}},{"id":"x-ai/grok-imagine-image-quality","object":"model","created":1779117584,"owned_by":"x-ai","canonical_slug":"x-ai/grok-imagine-image-quality-20260512","hugging_face_id":null,"name":"SpaceXAI: Grok Imagine Image Quality","description":"Grok Imagine Image Quality is SpaceXAI's fast, high-fidelity image generation and editing model. It accepts text prompts and optional reference images, producing photorealistic outputs at 1K or 2K across a...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0.01","image_token":"0.0000119760479041916","image_output":"0.0000119760479041916"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","seed","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-imagine-image-quality"}},{"id":"mistralai/voxtral-mini-transcribe","object":"model","created":1778877024,"owned_by":"mistralai","canonical_slug":"mistralai/voxtral-mini-transcribe-2602","hugging_face_id":null,"name":"Mistral: Voxtral Mini Transcribe","description":"Voxtral Mini Transcribe is Mistral's speech-to-text model, derived from the Voxtral Mini family. It accepts audio input and returns transcribed text via the standard transcription API. Suited for transcribing meetings,...","context_length":16384,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00005","completion":"0"},"top_provider":{"context_length":16384,"max_completion_tokens":13107,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/voxtral-mini-transcribe"}},{"id":"x-ai/grok-voice-tts-1.0","object":"model","created":1778805456,"owned_by":"x-ai","canonical_slug":"x-ai/grok-voice-tts-1.0","hugging_face_id":null,"name":"SpaceXAI: Grok Voice TTS 1.0","description":"Grok Voice TTS 1.0 is a text-to-speech model from SpaceXAI. It converts text into spoken audio across 20+ languages with automatic language detection, and offers five built-in voices (Eve, Ara,...","context_length":15000,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0"},"top_provider":{"context_length":15000,"max_completion_tokens":13500,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":["eve","ara","rex","sal","leo"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-voice-tts-1.0"}},{"id":"qwen/qwen3-asr-flash-2026-02-10","object":"model","created":1778732776,"owned_by":"qwen","canonical_slug":"qwen/qwen3-asr-flash-2026-02-10","hugging_face_id":null,"name":"Qwen: Qwen3 ASR Flash","description":"Qwen3-ASR-Flash is Alibaba's automatic speech recognition service, built on the Qwen3-Omni foundation and trained on tens of millions of hours of multimodal speech data. The model handles 11 languages —...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000035","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-asr-flash-2026-02-10"}},{"id":"recraft/recraft-v4.1-pro-vector","object":"model","created":1778707395,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-pro-vector-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4.1 Pro Vector","description":"Recraft V4.1 Pro Vector is the vector (SVG) variant of Recraft V4.1 Pro, tuned for high aesthetics. It supports text and image inputs and produces higher-resolution SVG image output across...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000718562874251497","image_output":"0.0000718562874251497"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1-pro-vector"}},{"id":"recraft/recraft-v4.1-vector","object":"model","created":1778707392,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-vector-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4.1 Vector","description":"Recraft V4.1 Vector is the vector (SVG) variant of Recraft V4.1, tuned for high aesthetics. It supports text and image inputs and produces SVG image output across multiple aspect ratios,...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000191616766467066","image_output":"0.0000191616766467066"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1-vector"}},{"id":"recraft/recraft-v4.1-utility-pro","object":"model","created":1778707389,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-utility-pro-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4.1 Utility Pro","description":"Recraft V4.1 Utility Pro is a general-purpose image generation model from Recraft. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios — double...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000502994011976048","image_output":"0.0000502994011976048"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1-utility-pro"}},{"id":"recraft/recraft-v4.1-utility","object":"model","created":1778707387,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-utility-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4.1 Utility","description":"Recraft V4.1 Utility is a general-purpose image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios, with typical generation...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000838323353293413","image_output":"0.00000838323353293413"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1-utility"}},{"id":"recraft/recraft-v4.1-pro","object":"model","created":1778707384,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-pro-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4.1 Pro","description":"Recraft V4.1 Pro is an image generation model from Recraft tuned for high aesthetics. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000502994011976048","image_output":"0.0000502994011976048"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1-pro"}},{"id":"recraft/recraft-v4.1","object":"model","created":1778707381,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4.1-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4.1","description":"Recraft V4.1 is an image generation model from Recraft tuned for high aesthetics. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios, with...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000838323353293413","image_output":"0.00000838323353293413"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4.1"}},{"id":"recraft/recraft-v4-pro-vector","object":"model","created":1778707334,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-pro-vector-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4 Pro Vector","description":"Recraft V4 Pro Vector is the vector (SVG) variant of Recraft V4 Pro. It supports text and image inputs and produces vector image output across multiple aspect ratios at the...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000718562874251497","image_output":"0.0000718562874251497"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-pro-vector"}},{"id":"recraft/recraft-v4-vector","object":"model","created":1778707333,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-vector-20260514","hugging_face_id":null,"name":"Recraft: Recraft V4 Vector","description":"Recraft V4 Vector is the vector (SVG) variant of Recraft V4. It supports text and image inputs and produces vector image output across multiple aspect ratios. Compared to the raster...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000191616766467066","image_output":"0.0000191616766467066"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-vector"}},{"id":"perceptron/perceptron-mk1","object":"model","created":1778597029,"owned_by":"perceptron","canonical_slug":"perceptron/perceptron-mk1-20260512","hugging_face_id":null,"name":"Perceptron: Perceptron Mk1","description":"Perceptron Mk1 (Mark One) is Perceptron's highest-quality vision-language model for video and embodied reasoning.** It accepts image and video inputs paired with natural language queries, and produces detailed visual understanding...","context_length":32768,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000015"},"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perceptron/perceptron-mk1"}},{"id":"recraft/recraft-v4-pro","object":"model","created":1778185441,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-pro-20260413","hugging_face_id":null,"name":"Recraft: Recraft V4 Pro","description":"Recraft V4 Pro is an image generation model from Recraft. It supports text and image inputs with image output at ~2K resolution across multiple aspect ratios, double the resolution of...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000598802395209581","image_output":"0.0000598802395209581"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4-pro"}},{"id":"recraft/recraft-v4","object":"model","created":1778185437,"owned_by":"recraft","canonical_slug":"recraft/recraft-v4-20260413","hugging_face_id":null,"name":"Recraft: Recraft V4","description":"Recraft V4 is an image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios. It delivers stronger compositional judgment,...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000958083832335329","image_output":"0.00000958083832335329"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v4"}},{"id":"recraft/recraft-v3","object":"model","created":1778185433,"owned_by":"recraft","canonical_slug":"recraft/recraft-v3-20260413","hugging_face_id":null,"name":"Recraft: Recraft V3","description":"Recraft V3 is an image generation model from Recraft. It supports text and image inputs with image output at ~1K resolution across multiple aspect ratios. Supports the following `image_config` parameters:...","context_length":65536,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000958083832335329","image_output":"0.00000958083832335329"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/recraft/recraft-v3"}},{"id":"google/gemini-3.1-flash-lite","object":"model","created":1778168828,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-lite-20260507","hugging_face_id":null,"name":"Google: Gemini 3.1 Flash Lite","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-lite"}},{"id":"google/gemini-3.1-flash-lite:batch","object":"model","created":1778168828,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-lite-20260507","hugging_face_id":null,"name":"Google: Gemini 3.1 Flash Lite (batch)","description":"Gemini 3.1 Flash Lite is Google’s GA high-efficiency multimodal model optimized for low-latency, high-volume workloads. It supports text, image, video, audio, and PDF inputs, and is designed for lightweight agentic...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000000125","completion":"0.00000075","image":"0.000000125","audio":"0.00000025","input_audio_cache":"0.000000025","web_search":"0.014","internal_reasoning":"0.00000075","input_cache_read":"0.0000000125"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-lite:batch"}},{"id":"openai/gpt-chat-latest","object":"model","created":1778000212,"owned_by":"openai","canonical_slug":"openai/gpt-chat-latest-20260505","hugging_face_id":null,"name":"OpenAI: GPT Chat Latest","description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-chat-latest"}},{"id":"google/chirp-3","object":"model","created":1777997783,"owned_by":"google","canonical_slug":"google/chirp-3","hugging_face_id":null,"name":"Google: Chirp 3","description":"Chirp 3 is Google's latest multilingual speech-to-text model. It offers enhanced transcription accuracy across 24 GA languages and 77+ preview languages, with support for automatic language detection, automatic punctuation, and...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000266666666667","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/chirp-3"}},{"id":"openai/gpt-4o-mini-transcribe","object":"model","created":1777658151,"owned_by":"openai","canonical_slug":"openai/gpt-4o-mini-transcribe","hugging_face_id":null,"name":"OpenAI: GPT-4o Mini Transcribe","description":"GPT-4o Mini Transcribe is OpenAI's smaller, cost-efficient speech-to-text model built on GPT-4o Mini audio capabilities. It's priced per token (input and output), making it suitable for high-volume transcription workflows that...","context_length":128000,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.000005"},"top_provider":{"context_length":128000,"max_completion_tokens":115200,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-mini-transcribe"}},{"id":"openai/whisper-large-v3","object":"model","created":1777642266,"owned_by":"openai","canonical_slug":"openai/whisper-large-v3","hugging_face_id":"openai/whisper-large-v3","name":"OpenAI: Whisper Large V3","description":"Whisper Large V3 is OpenAI's open-source automatic speech recognition model offering both audio transcription and translation. It supports 99+ languages and accepts common audio formats including mp3, mp4, wav, webm,...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000075","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/whisper-large-v3"}},{"id":"openai/whisper-large-v3-turbo","object":"model","created":1777642266,"owned_by":"openai","canonical_slug":"openai/whisper-large-v3-turbo","hugging_face_id":"openai/whisper-large-v3-turbo","name":"OpenAI: Whisper Large V3 Turbo","description":"Whisper Large V3 Turbo is an optimized version of OpenAI's Whisper Large V3 speech recognition model, designed for speed and cost efficiency. It supports transcription across 99+ languages with a...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000333","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/whisper-large-v3-turbo"}},{"id":"x-ai/grok-4.3","object":"model","created":1777591821,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.3-20260430","hugging_face_id":null,"name":"SpaceXAI: Grok 4.3","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":900000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.3"}},{"id":"x-ai/grok-4.3:batch","object":"model","created":1777591821,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.3-20260430","hugging_face_id":null,"name":"SpaceXAI: Grok 4.3 (batch)","description":"Grok 4.3 is a reasoning model from SpaceXAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000002","web_search":"0.005","input_cache_read":"0.00000016","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000004","input_cache_read":"0.00000032"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":900000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.3:batch"}},{"id":"mistralai/mistral-medium-3-5","object":"model","created":1777570439,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-medium-3.5-20260430","hugging_face_id":null,"name":"Mistral: Mistral Medium 3.5","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000015","completion":"0.0000075"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-medium-3-5"}},{"id":"mistralai/mistral-medium-3-5:batch","object":"model","created":1777570439,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-medium-3.5-20260430","hugging_face_id":null,"name":"Mistral: Mistral Medium 3.5 (batch)","description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-medium-3-5:batch"}},{"id":"kwaivgi/kling-v3.0-pro","object":"model","created":1777496206,"owned_by":"kwaivgi","canonical_slug":"kwaivgi/kling-v3.0-pro-20260429","hugging_face_id":null,"name":"Kling: Video v3.0 Pro","description":"Kling v3.0 Pro is Kuaishou's premium video generation model, offering higher visual quality than the Standard tier. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for precise...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/kwaivgi/kling-v3.0-pro"}},{"id":"kwaivgi/kling-v3.0-std","object":"model","created":1777496205,"owned_by":"kwaivgi","canonical_slug":"kwaivgi/kling-v3.0-std-20260429","hugging_face_id":null,"name":"Kling: Video v3.0 Standard","description":"Kling v3.0 Standard is a video generation model from Kuaishou. It supports text-to-video and image-to-video workflows, with first-frame and last-frame control for guided scene composition. Clips range from 3 to...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/kwaivgi/kling-v3.0-std"}},{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","object":"model","created":1777393095,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428","hugging_face_id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","name":"NVIDIA: Nemotron 3 Nano Omni (free)","description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","context_length":256000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":256000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free"}},{"id":"openai/whisper-1","object":"model","created":1777332905,"owned_by":"openai","canonical_slug":"openai/whisper-1","hugging_face_id":null,"name":"OpenAI: Whisper 1","description":"Whisper is OpenAI's open-source automatic speech recognition model, available via API as `whisper-1`. It supports transcription and translation across 50+ languages from audio files up to 25 MB. Accepts formats...","context_length":0,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0001","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":[],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/whisper-1"}},{"id":"openai/gpt-4o-transcribe","object":"model","created":1777332895,"owned_by":"openai","canonical_slug":"openai/gpt-4o-transcribe","hugging_face_id":null,"name":"OpenAI: GPT-4o Transcribe","description":"GPT-4o Transcribe is OpenAI's high-quality speech-to-text model built on GPT-4o audio capabilities. It's priced per token (input and output), making it suitable for workflows that benefit from token-level billing transparency.","context_length":128000,"architecture":{"modality":"audio->transcription","input_modalities":["audio"],"output_modalities":["transcription"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":128000,"max_completion_tokens":115200,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-transcribe"}},{"id":"~anthropic/claude-haiku-latest","object":"model","created":1777318492,"owned_by":"~anthropic","canonical_slug":"~anthropic/claude-haiku-latest","hugging_face_id":null,"name":"Anthropic: Claude Haiku Latest","description":"This model always redirects to the latest model in the Claude Haiku family.","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~anthropic/claude-haiku-latest"}},{"id":"~openai/gpt-mini-latest","object":"model","created":1777318471,"owned_by":"~openai","canonical_slug":"~openai/gpt-mini-latest","hugging_face_id":null,"name":"OpenAI: GPT Mini Latest","description":"This model always redirects to the latest model in the GPT Mini family.","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000045","web_search":"0.01","input_cache_read":"0.000000075"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/v1/models/~openai/gpt-mini-latest"}},{"id":"~google/gemini-pro-latest","object":"model","created":1777318451,"owned_by":"~google","canonical_slug":"~google/gemini-pro-latest","hugging_face_id":null,"name":"Google: Gemini Pro Latest","description":"This model always redirects to the latest model in the Gemini Pro family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000018","audio":"0.000004","input_audio_cache":"0.0000004","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~google/gemini-pro-latest"}},{"id":"~moonshotai/kimi-latest","object":"model","created":1777318428,"owned_by":"~moonshotai","canonical_slug":"~moonshotai/kimi-latest","hugging_face_id":null,"name":"MoonshotAI: Kimi Latest","description":"This model always redirects to the latest model in the Kimi family.","context_length":1048576,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000001125","completion":"0.0000112","input_cache_read":"0.00000029"},"top_provider":{"context_length":1048576,"max_completion_tokens":943718,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~moonshotai/kimi-latest"}},{"id":"~google/gemini-flash-latest","object":"model","created":1777318398,"owned_by":"~google","canonical_slug":"~google/gemini-flash-latest","hugging_face_id":null,"name":"Google: Gemini Flash Latest","description":"This model always redirects to the latest model in the Gemini Flash family.","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.00000375","image":"0.00000075","audio":"0.00000075","input_audio_cache":"0.000000075","web_search":"0.014","internal_reasoning":"0.00000375","input_cache_read":"0.000000075","input_cache_write":"0.0000000416666666666667"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~google/gemini-flash-latest"}},{"id":"~anthropic/claude-sonnet-latest","object":"model","created":1777318368,"owned_by":"~anthropic","canonical_slug":"~anthropic/claude-sonnet-latest","hugging_face_id":null,"name":"Anthropic: Claude Sonnet Latest","description":"This model always redirects to the latest model in the Claude Sonnet family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.00001","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.0000025","input_cache_write_1h":"0.000004"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~anthropic/claude-sonnet-latest"}},{"id":"qwen/qwen3.5-plus-20260420","object":"model","created":1777261368,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-plus-20260420","hugging_face_id":null,"name":"Qwen: Qwen3.5 Plus 2026-04-20","description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000018","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":256000,"prompt":"0.000000375","completion":"0.00000225","input_cache_write":"0.00000046875"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-plus-20260420"}},{"id":"qwen/qwen3.6-flash","object":"model","created":1777261362,"owned_by":"qwen","canonical_slug":"qwen/qwen3.6-flash","hugging_face_id":null,"name":"Qwen: Qwen3.6 Flash","description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001875","completion":"0.000001125","input_cache_write":"0.000000234375","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000075","completion":"0.000003","input_cache_write":"0.0000009375"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.6-flash"}},{"id":"qwen/qwen3.6-35b-a3b","object":"model","created":1777260255,"owned_by":"qwen","canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","hugging_face_id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen: Qwen3.6 35B A3B","description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.6-35b-a3b"}},{"id":"qwen/qwen3.6-max-preview","object":"model","created":1777260242,"owned_by":"qwen","canonical_slug":"qwen/qwen3.6-max-preview-20260420","hugging_face_id":null,"name":"Qwen: Qwen3.6 Max Preview","description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000001027","completion":"0.000006162","input_cache_write":"0.00000128375","overrides":[{"min_prompt_tokens":128000,"prompt":"0.00000158","completion":"0.00000948","input_cache_write":"0.000001975"}]},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3.6-max-preview"}},{"id":"qwen/qwen3.6-27b","object":"model","created":1777255064,"owned_by":"qwen","canonical_slug":"qwen/qwen3.6-27b-20260422","hugging_face_id":"Qwen/Qwen3.6-27B","name":"Qwen: Qwen3.6 27B","description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000032","completion":"0.0000032"},"top_provider":{"context_length":262144,"max_completion_tokens":81920,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.6-27b"}},{"id":"openai/gpt-5.5-pro","object":"model","created":1777051896,"owned_by":"openai","canonical_slug":"openai/gpt-5.5-pro-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5 Pro","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00018","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00006","completion":"0.00027"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.5-pro"}},{"id":"openai/gpt-5.5-pro:batch","object":"model","created":1777051896,"owned_by":"openai","canonical_slug":"openai/gpt-5.5-pro-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5 Pro (batch)","description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.00009","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00003","completion":"0.000135"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.5-pro:batch"}},{"id":"openai/gpt-5.5","object":"model","created":1777051893,"owned_by":"openai","canonical_slug":"openai/gpt-5.5-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.00003","web_search":"0.01","input_cache_read":"0.0000005","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00001","completion":"0.000045","input_cache_read":"0.000001"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.5"}},{"id":"openai/gpt-5.5:batch","object":"model","created":1777051893,"owned_by":"openai","canonical_slug":"openai/gpt-5.5-20260423","hugging_face_id":"","name":"OpenAI: GPT-5.5 (batch)","description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000005","completion":"0.0000225","input_cache_read":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-12-01","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.5:batch"}},{"id":"deepseek/deepseek-v4-pro","object":"model","created":1777000679,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4-pro-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek: DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000002088","completion":"0.0000004176","input_cache_read":"0.0000000174"},"top_provider":{"context_length":1024000,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4-pro"}},{"id":"deepseek/deepseek-v4-flash","object":"model","created":1777000666,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v4-flash-20260423","hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek: DeepSeek V4 Flash 0423","description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","context_length":1048576,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.000000042","completion":"0.000000084","input_cache_read":"0.0000000084"},"top_provider":{"context_length":1024000,"max_completion_tokens":384000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v4-flash"}},{"id":"google/gemini-3.1-flash-tts-preview","object":"model","created":1776999308,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-tts-preview","hugging_face_id":null,"name":"Google: Gemini 3.1 Flash TTS Preview","description":"Gemini 3.1 Flash TTS Preview is a text-to-speech model from Google, and a substantial generational step up from Gemini 2.5 Flash TTS. It takes text input and produces audio output...","context_length":32768,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.00002"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null},"supported_voices":["Zephyr","Puck","Charon","Kore","Fenrir","Leda","Orus","Aoede","Callirrhoe","Autonoe","Enceladus","Iapetus","Umbriel","Algieba","Despina","Erinome","Algenib","Rasalgethi","Laomedeia","Achernar","Alnilam","Schedar","Gacrux","Pulcherrima","Achird","Zubenelgenubi","Vindemiatrix","Sadachbia","Sadaltager","Sulafat"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-tts-preview"}},{"id":"google/veo-3.1-fast","object":"model","created":1776994666,"owned_by":"google","canonical_slug":"google/veo-3.1-fast-20260320","hugging_face_id":null,"name":"Google: Veo 3.1 Fast","description":"Google's mid-tier video generation model balancing speed and quality. Veo 3.1 Fast generates high-quality video from text or image prompts with native synchronized audio, offering faster turnaround than Veo 3.1...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/veo-3.1-fast"}},{"id":"canopylabs/orpheus-3b-0.1-ft","object":"model","created":1776983168,"owned_by":"canopylabs","canonical_slug":"canopylabs/orpheus-3b-0.1-ft","hugging_face_id":null,"name":"Canopy Labs: Orpheus 3B","description":"Orpheus 3B is an English text-to-speech model from Canopy Labs, fine-tuned for natural prosody and expressive delivery. It offers 7 preset voices and is suited for narration, voice assistants, and...","context_length":4096,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000007","completion":"0"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["tara","leah","jess","leo","dan","mia","zac"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/canopylabs/orpheus-3b-0.1-ft"}},{"id":"sesame/csm-1b","object":"model","created":1776983168,"owned_by":"sesame","canonical_slug":"sesame/csm-1b","hugging_face_id":null,"name":"Sesame: CSM 1B","description":"CSM 1B is a conversational speech model from Sesame. It accepts text input and produces English speech output, with voice options spanning conversational and read-speech styles. At 1B parameters, it...","context_length":4096,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000007","completion":"0"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["conversational_a","conversational_b","read_speech_a","read_speech_b","read_speech_c","read_speech_d","none"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sesame/csm-1b"}},{"id":"hexgrad/kokoro-82m","object":"model","created":1776983167,"owned_by":"hexgrad","canonical_slug":"hexgrad/kokoro-82m","hugging_face_id":null,"name":"hexgrad: Kokoro 82M","description":"Kokoro 82M is a lightweight, open-weight text-to-speech model from hexgrad. It converts text to speech across 8 languages (American and British English, Spanish, French, Hindi, Italian, Japanese, Portuguese, and Chinese)...","context_length":4096,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000062","completion":"0"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{},"supported_voices":["af_alloy","af_aoede","af_bella","af_heart","af_jessica","af_kore","af_nicole","af_nova","af_river","af_sarah","af_sky","am_adam","am_echo","am_eric","am_fenrir","am_liam","am_michael","am_onyx","am_puck","am_santa","bf_alice","bf_emma","bf_isabella","bf_lily","bm_daniel","bm_fable","bm_george","bm_lewis","ef_dora","em_alex","em_santa","ff_siwis","hf_alpha","hf_beta","hm_omega","hm_psi","if_sara","im_nicola","jf_alpha","jf_gongitsune","jf_nezumi","jf_tebukuro","jm_kumo","pf_dora","pm_alex","pm_santa","zf_xiaobei","zf_xiaoni","zf_xiaoxiao","zf_xiaoyi","zm_yunjian","zm_yunxi","zm_yunxia","zm_yunyang"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/hexgrad/kokoro-82m"}},{"id":"google/veo-3.1-lite","object":"model","created":1776978818,"owned_by":"google","canonical_slug":"google/veo-3.1-lite-20260331","hugging_face_id":null,"name":"Google: Veo 3.1 Lite","description":"Google's most cost-effective video generation model, designed for high-volume applications and rapid iteration. Veo 3.1 Lite generates 720p and 1080p video from text or image prompts with native synchronized audio...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/veo-3.1-lite"}},{"id":"tencent/hy3-preview","object":"model","created":1776878150,"owned_by":"tencent","canonical_slug":"tencent/hy3-preview-20260421","hugging_face_id":"tencent/Hy3-preview","name":"Tencent: Hy3 preview","description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000018","completion":"0.0000006","input_cache_read":"0.00000006"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.9,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/tencent/hy3-preview"}},{"id":"xiaomi/mimo-v2.5-pro","object":"model","created":1776874273,"owned_by":"xiaomi","canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","hugging_face_id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"Xiaomi: MiMo-V2.5-Pro","description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","context_length":1050000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000435","completion":"0.00000087","input_cache_read":"0.0000000036"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/xiaomi/mimo-v2.5-pro"}},{"id":"xiaomi/mimo-v2.5","object":"model","created":1776874269,"owned_by":"xiaomi","canonical_slug":"xiaomi/mimo-v2.5-20260422","hugging_face_id":"XiaomiMiMo/MiMo-V2.5","name":"Xiaomi: MiMo-V2.5","description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","context_length":1050000,"architecture":{"modality":"text+image+audio+video->text","input_modalities":["text","audio","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.00000028","input_cache_read":"0.0000000028"},"top_provider":{"context_length":1048576,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/xiaomi/mimo-v2.5"}},{"id":"openai/gpt-5.4-image-2","object":"model","created":1776797528,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-image-2-20260421","hugging_face_id":"","name":"OpenAI: GPT-5.4 Image 2","description":"GPT-5.4 Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","context_length":272000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000008","completion":"0.000015","image_output":"0.00003","web_search":"0.01","input_cache_read":"0.000002"},"top_provider":{"context_length":272000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","top_logprobs","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-image-2"}},{"id":"~anthropic/claude-opus-latest","object":"model","created":1776795361,"owned_by":"~anthropic","canonical_slug":"~anthropic/claude-opus-latest","hugging_face_id":"","name":"Anthropic: Claude Opus Latest","description":"This model always redirects to the latest model in the Claude Opus family.","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Router","instruct_type":null},"pricing":{"prompt":"0.000004","completion":"0.00002","web_search":"0.01","input_cache_read":"0.0000002","input_cache_write":"0.000005","input_cache_write_1h":"0.000008"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/~anthropic/claude-opus-latest"}},{"id":"kwaivgi/kling-video-o1","object":"model","created":1776704777,"owned_by":"kwaivgi","canonical_slug":"kwaivgi/kling-video-o1-20260420","hugging_face_id":null,"name":"Kling: Video O1","description":"Kling Video O1 is a video generation model from Kuaishou. It supports text and image inputs with video output, enabling text-to-video and image-to-video workflows. It is suited for cinematic content...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/kwaivgi/kling-video-o1"}},{"id":"minimax/hailuo-2.3","object":"model","created":1776702740,"owned_by":"minimax","canonical_slug":"minimax/hailuo-2.3-20260420","hugging_face_id":null,"name":"MiniMax: Hailuo 2.3","description":"Hailuo 2.3 is a video generation model from MiniMax. It accepts text prompts and reference images as input and generates video output, supporting both text-to-video and image-to-video workflows. It is...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/hailuo-2.3"}},{"id":"moonshotai/kimi-k2.6","object":"model","created":1776699402,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k2.6-20260420","hugging_face_id":"moonshotai/Kimi-K2.6","name":"MoonshotAI: Kimi K2.6","description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000043415","completion":"0.000001828","input_cache_read":"0.00000007312"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k2.6"}},{"id":"mistralai/voxtral-mini-tts-2603","object":"model","created":1776571337,"owned_by":"mistralai","canonical_slug":"mistralai/voxtral-mini-tts-2603","hugging_face_id":null,"name":"Mistral: Voxtral Mini TTS","description":"Voxtral Mini TTS is Mistral's text-to-speech model featuring zero-shot voice cloning and multilingual support. It converts text input into natural-sounding audio output.","context_length":4096,"architecture":{"modality":"text->speech","input_modalities":["text"],"output_modalities":["speech"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000016","completion":"0"},"top_provider":{"context_length":4096,"max_completion_tokens":3276,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":["en_paul_sad","en_paul_neutral","en_paul_happy","en_paul_frustrated","en_paul_excited","en_paul_confident","en_paul_cheerful","en_paul_angry","gb_oliver_neutral","gb_oliver_sad","gb_oliver_excited","gb_oliver_curious","gb_oliver_confident","gb_oliver_cheerful","gb_oliver_angry","gb_jane_sarcasm","gb_jane_confused","gb_jane_shameful","gb_jane_sad","gb_jane_neutral","gb_jane_jealousy","gb_jane_frustrated","gb_jane_curious","gb_jane_confident","fr_marie_sad","fr_marie_neutral","fr_marie_happy","fr_marie_excited","fr_marie_curious","fr_marie_angry"],"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/voxtral-mini-tts-2603"}},{"id":"google/gemini-embedding-2-preview","object":"model","created":1776436465,"owned_by":"google","canonical_slug":"google/gemini-embedding-2-preview","hugging_face_id":null,"name":"Google: Gemini Embedding 2 Preview","description":"Gemini Embedding 2 Preview is Google's first multimodal embedding model. We currently support mapping text and images into a unified vector space for semantic search and retrieval-augmented generation (RAG). It...","context_length":8192,"architecture":{"modality":"text+image+file+audio+video->embeddings","input_modalities":["text","image","file","audio","video"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0","image":"0.00000045","audio":"0.0000065"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-embedding-2-preview"}},{"id":"anthropic/claude-opus-4.7","object":"model","created":1776351100,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.7-opus-20260416","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.7","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.7"}},{"id":"anthropic/claude-opus-4.7:batch","object":"model","created":1776351100,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.7-opus-20260416","hugging_face_id":null,"name":"Anthropic: Claude Opus 4.7 (batch)","description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.000003125","input_cache_write_1h":"0.000005"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.7:batch"}},{"id":"alibaba/wan-2.7","object":"model","created":1776211362,"owned_by":"alibaba","canonical_slug":"alibaba/wan-2.7-20260414","hugging_face_id":null,"name":"Alibaba: Wan 2.7","description":"Wan 2.7 is a video generation model from Alibaba. It supports text-to-video, image-to-video with first and last frame control, and reference-to-video, where multiple reference images guide the style and content...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/alibaba/wan-2.7"}},{"id":"bytedance/seedance-2.0","object":"model","created":1776211362,"owned_by":"bytedance","canonical_slug":"bytedance/seedance-2.0-20260414","hugging_face_id":null,"name":"ByteDance: Seedance 2.0","description":"Seedance 2.0 is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It is particularly strong at preserving character consistency,...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance/seedance-2.0"}},{"id":"bytedance/seedance-2.0-fast","object":"model","created":1776211362,"owned_by":"bytedance","canonical_slug":"bytedance/seedance-2.0-fast-20260414","hugging_face_id":null,"name":"ByteDance: Seedance 2.0 Fast","description":"Seedance 2.0 Fast is a video generation model from ByteDance. It supports text-to-video, image-to-video with first and last frame control, and multimodal reference-to-video. It prioritizes generation speed and lower cost...","context_length":0,"architecture":{"modality":"text+image+audio+video->video","input_modalities":["text","image","video","audio"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance/seedance-2.0-fast"}},{"id":"z-ai/glm-5.1","object":"model","created":1775578025,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5.1-20260406","hugging_face_id":"zai-org/GLM-5.1","name":"Z.ai: GLM 5.1","description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000009646","completion":"0.0000030316","input_cache_read":"0.00000017914"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5.1"}},{"id":"cohere/rerank-4-pro","object":"model","created":1775446247,"owned_by":"cohere","canonical_slug":"cohere/rerank-4-pro","hugging_face_id":null,"name":"Cohere: Rerank 4 Pro","description":"Cohere's AI search foundation model for enhancing the relevance of information surfaced within search and RAG systems. Features a 32K context window, multilingual support across 100+ languages, no data pre-processing...","context_length":32768,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/cohere/rerank-4-pro"}},{"id":"cohere/rerank-4-fast","object":"model","created":1775442269,"owned_by":"cohere","canonical_slug":"cohere/rerank-4-fast","hugging_face_id":null,"name":"Cohere: Rerank 4 Fast","description":"Cohere's AI search foundation model for enhancing the relevance of information surfaced within search and RAG systems. Features a 32K context window, multilingual support across 100+ languages, no data pre-processing...","context_length":32768,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/cohere/rerank-4-fast"}},{"id":"cohere/rerank-v3.5","object":"model","created":1775416158,"owned_by":"cohere","canonical_slug":"cohere/rerank-v3.5","hugging_face_id":null,"name":"Cohere: Rerank v3.5","description":"Rerank v3.5 is designed to reorder search results for improved relevance. It supports multi-aspect and semi-structured data reranking over 100+ languages. Ideal for refining results from semantic or keyword search...","context_length":4096,"architecture":{"modality":"text->rerank","input_modalities":["text"],"output_modalities":["rerank"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/cohere/rerank-v3.5"}},{"id":"google/gemma-4-26b-a4b-it","object":"model","created":1775227989,"owned_by":"google","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","hugging_face_id":"google/gemma-4-26B-A4B-it","name":"Google: Gemma 4 26B A4B ","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.0000000765","completion":"0.000000255","input_cache_read":"0.0000000425"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemma-4-26b-a4b-it"}},{"id":"google/gemma-4-26b-a4b-it:free","object":"model","created":1775227989,"owned_by":"google","canonical_slug":"google/gemma-4-26b-a4b-it-20260403","hugging_face_id":"google/gemma-4-26B-A4B-it","name":"Google: Gemma 4 26B A4B  (free)","description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemma-4-26b-a4b-it:free"}},{"id":"google/gemma-4-31b-it","object":"model","created":1775148486,"owned_by":"google","canonical_slug":"google/gemma-4-31b-it-20260402","hugging_face_id":"google/gemma-4-31B-it","name":"Google: Gemma 4 31B","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0.00000009","completion":"0.00000034","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemma-4-31b-it"}},{"id":"google/gemma-4-31b-it:free","object":"model","created":1775148486,"owned_by":"google","canonical_slug":"google/gemma-4-31b-it-20260402","hugging_face_id":"google/gemma-4-31B-it","name":"Google: Gemma 4 31B (free)","description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Gemma","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":64,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemma-4-31b-it:free"}},{"id":"qwen/qwen3.6-plus","object":"model","created":1775133557,"owned_by":"qwen","canonical_slug":"qwen/qwen3.6-plus-04-02","hugging_face_id":"","name":"Qwen: Qwen3.6 Plus","description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000325","completion":"0.00000195","input_cache_write":"0.00000040625","overrides":[{"min_prompt_tokens":256000,"prompt":"0.0000013","completion":"0.0000039","input_cache_write":"0.000001625"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.6-plus"}},{"id":"z-ai/glm-5v-turbo","object":"model","created":1775061458,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5v-turbo-20260401","hugging_face_id":"","name":"Z.ai: GLM 5V Turbo","description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","context_length":202752,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5v-turbo"}},{"id":"arcee-ai/trinity-large-thinking","object":"model","created":1775058318,"owned_by":"arcee-ai","canonical_slug":"arcee-ai/trinity-large-thinking","hugging_face_id":"arcee-ai/Trinity-Large-Thinking","name":"Arcee AI: Trinity Large Thinking","description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000008","input_cache_read":"0.00000006"},"top_provider":{"context_length":262144,"max_completion_tokens":80000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.3,"top_p":0.8,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/arcee-ai/trinity-large-thinking"}},{"id":"x-ai/grok-4.20-multi-agent","object":"model","created":1774979158,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","hugging_face_id":"","name":"SpaceXAI: Grok 4.20 Multi-Agent","description":"Grok 4.20 Multi-Agent is a variant of SpaceXAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":2000000,"max_completion_tokens":1800000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-01","expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.20-multi-agent"}},{"id":"x-ai/grok-4.20","object":"model","created":1774979019,"owned_by":"x-ai","canonical_slug":"x-ai/grok-4.20-20260309","hugging_face_id":"","name":"SpaceXAI: Grok 4.20","description":"Grok 4.20 is a reasoning model from SpaceXAI with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering...","context_length":2000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Grok","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.0000025","web_search":"0.005","input_cache_read":"0.0000002","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000005","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":2000000,"max_completion_tokens":1800000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-01","expiration_date":null,"links":{"details":"/v1/models/x-ai/grok-4.20"}},{"id":"google/lyria-3-pro-preview","object":"model","created":1774907286,"owned_by":"google","canonical_slug":"google/lyria-3-pro-preview-20260330","hugging_face_id":null,"name":"Google: Lyria 3 Pro Preview","description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/lyria-3-pro-preview"}},{"id":"google/lyria-3-clip-preview","object":"model","created":1774907255,"owned_by":"google","canonical_slug":"google/lyria-3-clip-preview-20260330","hugging_face_id":null,"name":"Google: Lyria 3 Clip Preview","description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","context_length":1048576,"architecture":{"modality":"text+image->text+audio","input_modalities":["text","image"],"output_modalities":["text","audio"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/lyria-3-clip-preview"}},{"id":"alibaba/wan-2.6","object":"model","created":1774659190,"owned_by":"alibaba","canonical_slug":"alibaba/wan-2.6-20260327","hugging_face_id":null,"name":"Alibaba: Wan 2.6","description":"Alibaba's most advanced video generation model, supporting over 10 visual creation capabilities in a unified system. Wan 2.6 generates 1080p video at 24fps from text, images, reference videos, or audio,...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/alibaba/wan-2.6"}},{"id":"bytedance/seedance-1-5-pro","object":"model","created":1774277608,"owned_by":"bytedance","canonical_slug":"bytedance/seedance-1-5-pro-20260320","hugging_face_id":null,"name":"ByteDance: Seedance 1.5 Pro","description":"ByteDance's next-generation audio-visual generation model with a 4.5B parameter Dual-Branch Diffusion Transformer architecture. Seedance 1.5 Pro generates video and audio simultaneously in a single unified pass — eliminating the timing...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-11-11","links":{"details":"/v1/models/bytedance/seedance-1-5-pro"}},{"id":"openai/sora-2-pro","object":"model","created":1774277521,"owned_by":"openai","canonical_slug":"openai/sora-2-pro-20260320","hugging_face_id":null,"name":"OpenAI: Sora 2 Pro","description":"OpenAI's flagship video generation model, delivering production-quality video with physics-accurate motion, synchronized audio, and world-state persistence across shots. Sora 2 Pro follows intricate multi-shot instructions while maintaining consistent spatial relationships...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","presence_penalty","stop","top_logprobs"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/sora-2-pro"}},{"id":"google/veo-3.1","object":"model","created":1774277148,"owned_by":"google","canonical_slug":"google/veo-3.1-20260320","hugging_face_id":null,"name":"Google: Veo 3.1","description":"Google's state-of-the-art video generation model, built for maximum visual fidelity in final production cuts. Veo 3.1 generates high-quality 1080p video from text or image prompts with native synchronized audio —...","context_length":0,"architecture":{"modality":"text+image->video","input_modalities":["text","image"],"output_modalities":["video"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":0,"max_completion_tokens":0,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/veo-3.1"}},{"id":"rekaai/reka-edge","object":"model","created":1774026965,"owned_by":"rekaai","canonical_slug":"rekaai/reka-edge-2603","hugging_face_id":"RekaAI/reka-edge-2603","name":"Reka Edge","description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","context_length":16384,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000001"},"top_provider":{"context_length":16384,"max_completion_tokens":14745,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/rekaai/reka-edge"}},{"id":"minimax/minimax-m2.7","object":"model","created":1773836697,"owned_by":"minimax","canonical_slug":"minimax/minimax-m2.7-20260318","hugging_face_id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax: MiniMax M2.7","description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000021","completion":"0.00000084","input_cache_read":"0.000000042"},"top_provider":{"context_length":196608,"max_completion_tokens":176947,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-m2.7"}},{"id":"openai/gpt-5.4-nano","object":"model","created":1773748187,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-nano-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Nano","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.00000125","web_search":"0.01","input_cache_read":"0.00000002"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-nano"}},{"id":"openai/gpt-5.4-nano:batch","object":"model","created":1773748187,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-nano-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Nano (batch)","description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.000000625","web_search":"0.01","input_cache_read":"0.00000001"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-nano:batch"}},{"id":"openai/gpt-5.4-mini","object":"model","created":1773748178,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-mini-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Mini","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000075","completion":"0.0000045","web_search":"0.01","input_cache_read":"0.000000075"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-mini"}},{"id":"openai/gpt-5.4-mini:batch","object":"model","created":1773748178,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-mini-20260317","hugging_face_id":"","name":"OpenAI: GPT-5.4 Mini (batch)","description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000375","completion":"0.00000225","web_search":"0.01","input_cache_read":"0.0000000375"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-08-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-mini:batch"}},{"id":"mistralai/mistral-small-2603","object":"model","created":1773695685,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-small-2603","hugging_face_id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral: Mistral Small 4","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-small-2603"}},{"id":"mistralai/mistral-small-2603:batch","object":"model","created":1773695685,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-small-2603","hugging_face_id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral: Mistral Small 4 (batch)","description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003","input_cache_read":"0.0000000075"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-small-2603:batch"}},{"id":"perplexity/pplx-embed-v1-4b","object":"model","created":1773625372,"owned_by":"perplexity","canonical_slug":"perplexity/pplx-embed-v1-4B","hugging_face_id":"","name":"Perplexity: Embed V1 4B","description":"pplx-embed-v1 -4B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 4B parameter model maximizing retrieval...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000003","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/pplx-embed-v1-4b"}},{"id":"perplexity/pplx-embed-v1-0.6b","object":"model","created":1773624868,"owned_by":"perplexity","canonical_slug":"perplexity/pplx-embed-v1-0.6B","hugging_face_id":"","name":"Perplexity: Embed V1 0.6B","description":"pplx-embed-v1-0.6B is one of Perplexity's state-of-the-art text embedding models built for real-world, web-scale retrieval. pplx-embed-v1 is optimized for standard dense text retrieval with the 0.6B parameter model targeting lightweight, low-latency...","context_length":32000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000004","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/pplx-embed-v1-0.6b"}},{"id":"z-ai/glm-5-turbo","object":"model","created":1773583573,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5-turbo-20260315","hugging_face_id":"","name":"Z.ai: GLM 5 Turbo","description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","context_length":202752,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000012","completion":"0.000004","input_cache_read":"0.00000024"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5-turbo"}},{"id":"nvidia/nemotron-3-super-120b-a12b","object":"model","created":1773245239,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"NVIDIA: Nemotron 3 Super","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000008","completion":"0.00000045"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-super-120b-a12b"}},{"id":"nvidia/nemotron-3-super-120b-a12b:free","object":"model","created":1773245239,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"NVIDIA: Nemotron 3 Super (free)","description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-super-120b-a12b:free"}},{"id":"bytedance-seed/seed-2.0-lite","object":"model","created":1773157231,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","hugging_face_id":null,"name":"ByteDance Seed: Seed-2.0-Lite","description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seed-2.0-lite"}},{"id":"qwen/qwen3.5-9b","object":"model","created":1773152396,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-9b-20260310","hugging_face_id":"Qwen/Qwen3.5-9B","name":"Qwen: Qwen3.5-9B","description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.00000015"},"top_provider":{"context_length":256000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-9b"}},{"id":"openai/gpt-5.4-pro","object":"model","created":1772734366,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-pro-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4 Pro","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00018","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00006","completion":"0.00027"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-pro"}},{"id":"openai/gpt-5.4-pro:batch","object":"model","created":1772734366,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-pro-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4 Pro (batch)","description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.00009","web_search":"0.01","overrides":[{"min_prompt_tokens":272000,"prompt":"0.00003","completion":"0.000135"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4-pro:batch"}},{"id":"openai/gpt-5.4","object":"model","created":1772734352,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.000015","web_search":"0.01","input_cache_read":"0.00000025","overrides":[{"min_prompt_tokens":272000,"prompt":"0.000005","completion":"0.0000225","input_cache_read":"0.0000005"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4"}},{"id":"openai/gpt-5.4:batch","object":"model","created":1772734352,"owned_by":"openai","canonical_slug":"openai/gpt-5.4-20260305","hugging_face_id":"","name":"OpenAI: GPT-5.4 (batch)","description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","context_length":1050000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.0000075","web_search":"0.01","input_cache_read":"0.000000125","overrides":[{"min_prompt_tokens":272000,"prompt":"0.0000025","completion":"0.00001125","input_cache_read":"0.00000025"}]},"top_provider":{"context_length":1050000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.4:batch"}},{"id":"inception/mercury-2","object":"model","created":1772636275,"owned_by":"inception","canonical_slug":"inception/mercury-2-20260304","hugging_face_id":null,"name":"Inception: Mercury 2","description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.00000075","input_cache_read":"0.000000025"},"top_provider":{"context_length":128000,"max_completion_tokens":50000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"default_parameters":{"temperature":0.75,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/inception/mercury-2"}},{"id":"google/gemini-3.1-flash-lite-preview","object":"model","created":1772512673,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","hugging_face_id":"","name":"Google: Gemini 3.1 Flash Lite Preview","description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","video","file","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","input_audio_cache":"0.00000005","web_search":"0.014","internal_reasoning":"0.0000015","input_cache_read":"0.000000025","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-lite-preview"}},{"id":"bytedance-seed/seed-2.0-mini","object":"model","created":1772131107,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","hugging_face_id":"","name":"ByteDance Seed: Seed-2.0-Mini","description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000002","completion":"0.0000008"}]},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seed-2.0-mini"}},{"id":"google/gemini-3.1-flash-image-preview","object":"model","created":1772119558,"owned_by":"google","canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","hugging_face_id":"","name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","image_output":"0.00006","web_search":"0.014"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-flash-image-preview"}},{"id":"qwen/qwen3.5-35b-a3b","object":"model","created":1772053822,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","hugging_face_id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen: Qwen3.5-35B-A3B","description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.000001","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-35b-a3b"}},{"id":"qwen/qwen3.5-27b","object":"model","created":1772053810,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-27b-20260224","hugging_face_id":"Qwen/Qwen3.5-27B","name":"Qwen: Qwen3.5-27B","description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000195","completion":"0.00000156"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-27b"}},{"id":"qwen/qwen3.5-122b-a10b","object":"model","created":1772053789,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","hugging_face_id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen: Qwen3.5-122B-A10B","description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000208"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-122b-a10b"}},{"id":"qwen/qwen3.5-flash-02-23","object":"model","created":1772053776,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-flash-20260224","hugging_face_id":null,"name":"Qwen: Qwen3.5-Flash","description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000065","completion":"0.00000026"},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-flash-02-23"}},{"id":"google/gemini-3.1-pro-preview-customtools","object":"model","created":1772045923,"owned_by":"google","canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","hugging_face_id":null,"name":"Google: Gemini 3.1 Pro Preview Custom Tools","description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","audio","image","video","file"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000018","audio":"0.000004","input_audio_cache":"0.0000004","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-pro-preview-customtools"}},{"id":"nvidia/llama-nemotron-embed-vl-1b-v2:free","object":"model","created":1772045017,"owned_by":"nvidia","canonical_slug":"nvidia/llama-nemotron-embed-vl-1b-v2-20260224","hugging_face_id":"nvidia/llama-nemotron-embed-vl-1b-v2","name":"NVIDIA: Llama Nemotron Embed VL 1B V2 (free)","description":"The Llama Nemotron Embed VL 1B V2 embedding model is optimized for multimodal question-answering retrieval. The model can embed 'documents' in the form of image, text, or image and text...","context_length":131072,"architecture":{"modality":"text+image->embeddings","input_modalities":["text","image"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/llama-nemotron-embed-vl-1b-v2:free"}},{"id":"openai/gpt-5.3-codex","object":"model","created":1771959164,"owned_by":"openai","canonical_slug":"openai/gpt-5.3-codex-20260224","hugging_face_id":"","name":"OpenAI: GPT-5.3-Codex","description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.3-codex"}},{"id":"aion-labs/aion-2.0","object":"model","created":1771881306,"owned_by":"aion-labs","canonical_slug":"aion-labs/aion-2.0-20260223","hugging_face_id":null,"name":"AionLabs: Aion-2.0","description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000016","input_cache_read":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/aion-labs/aion-2.0"}},{"id":"google/gemini-3.1-pro-preview","object":"model","created":1771509627,"owned_by":"google","canonical_slug":"google/gemini-3.1-pro-preview-20260219","hugging_face_id":"","name":"Google: Gemini 3.1 Pro Preview","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000004","completion":"0.000018","audio":"0.000004","input_audio_cache":"0.0000004","input_cache_read":"0.0000004"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-pro-preview"}},{"id":"google/gemini-3.1-pro-preview:batch","object":"model","created":1771509627,"owned_by":"google","canonical_slug":"google/gemini-3.1-pro-preview-20260219","hugging_face_id":"","name":"Google: Gemini 3.1 Pro Preview (batch)","description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["audio","file","image","text","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000006","image":"0.000001","audio":"0.000001","web_search":"0.014","internal_reasoning":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000002","completion":"0.000009","audio":"0.000002"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3.1-pro-preview:batch"}},{"id":"anthropic/claude-sonnet-4.6","object":"model","created":1771342990,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.6","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-4.6"}},{"id":"anthropic/claude-sonnet-4.6:batch","object":"model","created":1771342990,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.6-sonnet-20260217","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.6 (batch)","description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000015","completion":"0.0000075","web_search":"0.01","input_cache_read":"0.00000015","input_cache_write":"0.000001875","input_cache_write_1h":"0.000003"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-4.6:batch"}},{"id":"qwen/qwen3.5-plus-02-15","object":"model","created":1771229416,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-plus-20260216","hugging_face_id":"","name":"Qwen: Qwen3.5 Plus 2026-02-15","description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","context_length":1000000,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000156","overrides":[{"min_prompt_tokens":256000,"prompt":"0.000000325","completion":"0.00000195"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-plus-02-15"}},{"id":"qwen/qwen3.5-397b-a17b","object":"model","created":1771223018,"owned_by":"qwen","canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","hugging_face_id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen: Qwen3.5 397B A17B","description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["text","image","video"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000035","input_cache_read":"0.000000225"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3.5-397b-a17b"}},{"id":"minimax/minimax-m2.5","object":"model","created":1770908502,"owned_by":"minimax","canonical_slug":"minimax/minimax-m2.5-20260211","hugging_face_id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax: MiniMax M2.5","description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000027","completion":"0.00000108","input_cache_read":"0.000000027"},"top_provider":{"context_length":200000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-m2.5"}},{"id":"z-ai/glm-5","object":"model","created":1770829182,"owned_by":"z-ai","canonical_slug":"z-ai/glm-5-20260211","hugging_face_id":"zai-org/GLM-5","name":"Z.ai: GLM 5","description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.00000192","input_cache_read":"0.00000012"},"top_provider":{"context_length":198000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-5"}},{"id":"qwen/qwen3-max-thinking","object":"model","created":1770671901,"owned_by":"qwen","canonical_slug":"qwen/qwen3-max-thinking-20260123","hugging_face_id":null,"name":"Qwen: Qwen3 Max Thinking","description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000078","completion":"0.0000039","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000156","completion":"0.0000078"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975"}]},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-max-thinking"}},{"id":"anthropic/claude-opus-4.6","object":"model","created":1770219050,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.6-opus-20260205","hugging_face_id":"","name":"Anthropic: Claude Opus 4.6","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.6"}},{"id":"anthropic/claude-opus-4.6:batch","object":"model","created":1770219050,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.6-opus-20260205","hugging_face_id":"","name":"Anthropic: Claude Opus 4.6 (batch)","description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.000003125","input_cache_write_1h":"0.000005"},"top_provider":{"context_length":1000000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.6:batch"}},{"id":"qwen/qwen3-coder-next","object":"model","created":1770164101,"owned_by":"qwen","canonical_slug":"qwen/qwen3-coder-next-2025-02-03","hugging_face_id":"Qwen/Qwen3-Coder-Next","name":"Qwen: Qwen3 Coder Next","description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000012","completion":"0.0000008","input_cache_read":"0.00000007"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-coder-next"}},{"id":"sourceful/riverflow-v2-pro","object":"model","created":1770051427,"owned_by":"sourceful","canonical_slug":"sourceful/riverflow-v2-pro-20260130","hugging_face_id":"","name":"Sourceful: Riverflow V2 Pro","description":"Riverflow V2 Pro is the most powerful variant of Sourceful's Riverflow 2.0 lineup, best for top-tier control and perfect text rendering. The Riverflow 2.0 series represents SOTA performance on image...","context_length":8192,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000359281437125749","image_output":"0.0000359281437125749"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sourceful/riverflow-v2-pro"}},{"id":"sourceful/riverflow-v2-fast","object":"model","created":1770051423,"owned_by":"sourceful","canonical_slug":"sourceful/riverflow-v2-fast-20260130","hugging_face_id":"","name":"Sourceful: Riverflow V2 Fast","description":"Riverflow V2 Fast is the fastest variant of Sourceful's Riverflow 2.0 lineup, best for production deployments and latency-critical workflows. The Riverflow 2.0 series represents SOTA performance on image generation and...","context_length":8192,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.00000479041916167665","image_output":"0.00000479041916167665"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sourceful/riverflow-v2-fast"}},{"id":"stepfun/step-3.5-flash","object":"model","created":1769728337,"owned_by":"stepfun","canonical_slug":"stepfun/step-3.5-flash","hugging_face_id":"stepfun-ai/Step-3.5-Flash","name":"StepFun: Step 3.5 Flash","description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/stepfun/step-3.5-flash"}},{"id":"moonshotai/kimi-k2.5","object":"model","created":1769487076,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k2.5-0127","hugging_face_id":"moonshotai/Kimi-K2.5","name":"MoonshotAI: Kimi K2.5","description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000045","completion":"0.00000225","input_cache_read":"0.00000007"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k2.5"}},{"id":"upstage/solar-pro-3","object":"model","created":1769481200,"owned_by":"upstage","canonical_slug":"upstage/solar-pro-3","hugging_face_id":"","name":"Upstage: Solar Pro 3","description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000015"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","response_format","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/upstage/solar-pro-3"}},{"id":"minimax/minimax-m2-her","object":"model","created":1769177239,"owned_by":"minimax","canonical_slug":"minimax/minimax-m2-her-20260123","hugging_face_id":"","name":"MiniMax: MiniMax M2-her","description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"top_provider":{"context_length":65536,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-m2-her"}},{"id":"writer/palmyra-x5","object":"model","created":1769003823,"owned_by":"writer","canonical_slug":"writer/palmyra-x5-20250428","hugging_face_id":"","name":"Writer: Palmyra X5","description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","context_length":1040000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.000006"},"top_provider":{"context_length":1040000,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/writer/palmyra-x5"}},{"id":"openai/gpt-audio","object":"model","created":1768862569,"owned_by":"openai","canonical_slug":"openai/gpt-audio","hugging_face_id":"","name":"OpenAI: GPT Audio","description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","audio":"0.000032","audio_output":"0.000064"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-audio"}},{"id":"openai/gpt-audio-mini","object":"model","created":1768859419,"owned_by":"openai","canonical_slug":"openai/gpt-audio-mini","hugging_face_id":"","name":"OpenAI: GPT Audio Mini","description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","context_length":128000,"architecture":{"modality":"text+audio->text+audio","input_modalities":["text","audio"],"output_modalities":["text","audio"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000024","audio":"0.0000006","audio_output":"0.0000024"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-audio-mini"}},{"id":"z-ai/glm-4.7-flash","object":"model","created":1768833913,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.7-flash-20260119","hugging_face_id":"zai-org/GLM-4.7-Flash","name":"Z.ai: GLM 4.7 Flash","description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","context_length":200000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000000605","completion":"0.0000004"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-4.7-flash"}},{"id":"black-forest-labs/flux.2-klein-4b","object":"model","created":1768429228,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux.2-klein-4b","hugging_face_id":"black-forest-labs/FLUX.2-klein-4B","name":"Black Forest Labs: FLUX.2 Klein 4B","description":"FLUX.2 [klein] 4B is the fastest and most cost-effective model in the FLUX.2 family, optimized for high-throughput use cases while maintaining excellent image quality. Pricing is based on the output...","context_length":40960,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_output":"0.00000341796875"},"top_provider":{"context_length":40960,"max_completion_tokens":36864,"is_moderated":false},"per_request_limits":null,"supported_parameters":["seed"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux.2-klein-4b"}},{"id":"openai/gpt-5.2-codex","object":"model","created":1768409315,"owned_by":"openai","canonical_slug":"openai/gpt-5.2-codex-20260114","hugging_face_id":"","name":"OpenAI: GPT-5.2-Codex","description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.2-codex"}},{"id":"bytedance-seed/seedream-4.5","object":"model","created":1766519506,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seedream-4.5-20251203","hugging_face_id":"","name":"ByteDance Seed: Seedream 4.5","description":"Seedream 4.5 is the latest in-house image generation model developed by ByteDance. Compared with Seedream 4.0, it delivers comprehensive improvements, especially in editing consistency, including better preservation of subject details,...","context_length":4096,"architecture":{"modality":"text+image->image","input_modalities":["image","text"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image":"0","image_token":"0.00000958083832335329","image_output":"0.00000958083832335329"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/bytedance-seed/seedream-4.5"}},{"id":"bytedance-seed/seed-1.6-flash","object":"model","created":1766505011,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","hugging_face_id":"","name":"ByteDance Seed: Seed 1.6 Flash","description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000001","completion":"0.0000008"}]},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-11-11","links":{"details":"/v1/models/bytedance-seed/seed-1.6-flash"}},{"id":"bytedance-seed/seed-1.6","object":"model","created":1766504997,"owned_by":"bytedance-seed","canonical_slug":"bytedance-seed/seed-1.6-20250625","hugging_face_id":"","name":"ByteDance Seed: Seed 1.6","description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","context_length":262144,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002","overrides":[{"min_prompt_tokens":128000,"prompt":"0.0000005","completion":"0.000004"}]},"top_provider":{"context_length":262144,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-11-11","links":{"details":"/v1/models/bytedance-seed/seed-1.6"}},{"id":"minimax/minimax-m2.1","object":"model","created":1766454997,"owned_by":"minimax","canonical_slug":"minimax/minimax-m2.1","hugging_face_id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax: MiniMax M2.1","description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012","input_cache_read":"0.00000003"},"top_provider":{"context_length":204800,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.9,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-08","links":{"details":"/v1/models/minimax/minimax-m2.1"}},{"id":"z-ai/glm-4.7","object":"model","created":1766378014,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.7-20251222","hugging_face_id":"zai-org/GLM-4.7","name":"Z.ai: GLM 4.7","description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"top_provider":{"context_length":202752,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-12-31","links":{"details":"/v1/models/z-ai/glm-4.7"}},{"id":"google/gemini-3-flash-preview","object":"model","created":1765987078,"owned_by":"google","canonical_slug":"google/gemini-3-flash-preview-20251217","hugging_face_id":"","name":"Google: Gemini 3 Flash Preview","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.000003","image":"0.0000005","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.000003","input_cache_read":"0.00000005","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3-flash-preview"}},{"id":"google/gemini-3-flash-preview:batch","object":"model","created":1765987078,"owned_by":"google","canonical_slug":"google/gemini-3-flash-preview-20251217","hugging_face_id":"","name":"Google: Gemini 3 Flash Preview (batch)","description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.0000015","image":"0.00000025","audio":"0.0000005","web_search":"0.014","internal_reasoning":"0.0000015"},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3-flash-preview:batch"}},{"id":"black-forest-labs/flux.2-max","object":"model","created":1765857570,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux.2-max","hugging_face_id":"","name":"Black Forest Labs: FLUX.2 Max","description":"FLUX.2 [max] is the new top-tier image model from Black Forest Labs, pushing image quality, prompt understanding, and editing consistency to the highest level yet. Pricing is as follows, [per...","context_length":46864,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_output":"0.00001708984375"},"top_provider":{"context_length":46864,"max_completion_tokens":42177,"is_moderated":false},"per_request_limits":null,"supported_parameters":["seed"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux.2-max"}},{"id":"nvidia/nemotron-3-nano-30b-a3b","object":"model","created":1765731275,"owned_by":"nvidia","canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","name":"NVIDIA: Nemotron 3 Nano 30B A3B","description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000002","input_cache_read":"0.00000003"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/nvidia/nemotron-3-nano-30b-a3b"}},{"id":"openai/gpt-5.2-chat","object":"model","created":1765389783,"owned_by":"openai","canonical_slug":"openai/gpt-5.2-chat-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2 Chat","description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"top_provider":{"context_length":128000,"max_completion_tokens":32000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_completion_tokens","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.2-chat"}},{"id":"openai/gpt-5.2-pro","object":"model","created":1765389780,"owned_by":"openai","canonical_slug":"openai/gpt-5.2-pro-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2 Pro","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000021","completion":"0.000168","web_search":"0.01"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.2-pro"}},{"id":"openai/gpt-5.2-pro:batch","object":"model","created":1765389780,"owned_by":"openai","canonical_slug":"openai/gpt-5.2-pro-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2 Pro (batch)","description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000105","completion":"0.000084","web_search":"0.01"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.2-pro:batch"}},{"id":"openai/gpt-5.2","object":"model","created":1765389775,"owned_by":"openai","canonical_slug":"openai/gpt-5.2-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000175","completion":"0.000014","web_search":"0.01","input_cache_read":"0.000000175"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.2"}},{"id":"openai/gpt-5.2:batch","object":"model","created":1765389775,"owned_by":"openai","canonical_slug":"openai/gpt-5.2-20251211","hugging_face_id":"","name":"OpenAI: GPT-5.2 (batch)","description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000875","completion":"0.000007","web_search":"0.01","input_cache_read":"0.0000000875"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.2:batch"}},{"id":"mistralai/devstral-2512","object":"model","created":1765285419,"owned_by":"mistralai","canonical_slug":"mistralai/devstral-2512","hugging_face_id":"mistralai/Devstral-2-123B-Instruct-2512","name":"Mistral: Devstral 2 2512","description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","context_length":262144,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/devstral-2512"}},{"id":"relace/relace-search","object":"model","created":1765213560,"owned_by":"relace","canonical_slug":"relace/relace-search-20251208","hugging_face_id":null,"name":"Relace: Relace Search","description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000003"},"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/relace/relace-search"}},{"id":"z-ai/glm-4.6v","object":"model","created":1765207462,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.6-20251208","hugging_face_id":"zai-org/GLM-4.6V","name":"Z.ai: GLM 4.6V","description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","context_length":131072,"architecture":{"modality":"text+image+video->text","input_modalities":["image","text","video"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000005"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.6,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-4.6v"}},{"id":"openai/gpt-5.1-codex-max","object":"model","created":1764878934,"owned_by":"openai","canonical_slug":"openai/gpt-5.1-codex-max-20251204","hugging_face_id":"","name":"OpenAI: GPT-5.1-Codex-Max","description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.1-codex-max"}},{"id":"amazon/nova-2-lite-v1","object":"model","created":1764696672,"owned_by":"amazon","canonical_slug":"amazon/nova-2-lite-v1","hugging_face_id":"","name":"Amazon: Nova 2 Lite","description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","context_length":1000000,"architecture":{"modality":"text+image+file+video->text","input_modalities":["text","image","video","file"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025"},"top_provider":{"context_length":1000000,"max_completion_tokens":65535,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/amazon/nova-2-lite-v1"}},{"id":"mistralai/ministral-14b-2512","object":"model","created":1764681735,"owned_by":"mistralai","canonical_slug":"mistralai/ministral-14b-2512","hugging_face_id":"mistralai/Ministral-3-14B-Instruct-2512","name":"Mistral: Ministral 3 14B 2512","description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000002","input_cache_read":"0.00000002"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/ministral-14b-2512"}},{"id":"mistralai/ministral-8b-2512","object":"model","created":1764681654,"owned_by":"mistralai","canonical_slug":"mistralai/ministral-8b-2512","hugging_face_id":"mistralai/Ministral-3-8B-Instruct-2512","name":"Mistral: Ministral 3 8B 2512","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000015","input_cache_read":"0.000000015"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/ministral-8b-2512"}},{"id":"mistralai/ministral-8b-2512:batch","object":"model","created":1764681654,"owned_by":"mistralai","canonical_slug":"mistralai/ministral-8b-2512","hugging_face_id":"mistralai/Ministral-3-8B-Instruct-2512","name":"Mistral: Ministral 3 8B 2512 (batch)","description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.000000075","input_cache_read":"0.0000000075"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/ministral-8b-2512:batch"}},{"id":"mistralai/ministral-3b-2512","object":"model","created":1764681560,"owned_by":"mistralai","canonical_slug":"mistralai/ministral-3b-2512","hugging_face_id":"mistralai/Ministral-3-3B-Instruct-2512","name":"Mistral: Ministral 3 3B 2512","description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000001","input_cache_read":"0.00000001"},"top_provider":{"context_length":131072,"max_completion_tokens":104857,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/ministral-3b-2512"}},{"id":"mistralai/mistral-large-2512","object":"model","created":1764624472,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-large-2512","hugging_face_id":"","name":"Mistral: Mistral Large 3 2512","description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000015","input_cache_read":"0.00000005"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.0645,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-large-2512"}},{"id":"mistralai/mistral-large-2512:batch","object":"model","created":1764624472,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-large-2512","hugging_face_id":"","name":"Mistral: Mistral Large 3 2512 (batch)","description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","context_length":262144,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.00000075","input_cache_read":"0.000000025"},"top_provider":{"context_length":262144,"max_completion_tokens":209715,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.0645,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-large-2512:batch"}},{"id":"deepseek/deepseek-v3.2","object":"model","created":1764594642,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v3.2-20251201","hugging_face_id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek: DeepSeek V3.2","description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000028","completion":"0.00000042","input_cache_read":"0.000000028"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v3.2"}},{"id":"black-forest-labs/flux.2-flex","object":"model","created":1764045987,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux.2-flex","hugging_face_id":"","name":"Black Forest Labs: FLUX.2 Flex","description":"FLUX.2 [flex] excels at rendering complex text, typography, and fine details, and supports multi-reference editing in the same unified architecture. Pricing is as follows, [per the docs](https://bfl.ai/pricing?category=flux.2): We charge $0.06...","context_length":67344,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_token":"0.0000146484375","image_output":"0.0000146484375"},"top_provider":{"context_length":67344,"max_completion_tokens":60609,"is_moderated":false},"per_request_limits":null,"supported_parameters":["seed"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux.2-flex"}},{"id":"black-forest-labs/flux.2-pro","object":"model","created":1764030274,"owned_by":"black-forest-labs","canonical_slug":"black-forest-labs/flux.2-pro","hugging_face_id":"","name":"Black Forest Labs: FLUX.2 Pro","description":"A high-end image generation and editing model focused on frontier-level visual quality and reliability. It delivers strong prompt adherence, stable lighting, sharp textures, and consistent character/style reproduction across multi-reference inputs....","context_length":46864,"architecture":{"modality":"text+image->image","input_modalities":["text","image"],"output_modalities":["image"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0","completion":"0","image_output":"0.00000732421875"},"top_provider":{"context_length":46864,"max_completion_tokens":42177,"is_moderated":false},"per_request_limits":null,"supported_parameters":["seed"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/black-forest-labs/flux.2-pro"}},{"id":"anthropic/claude-opus-4.5","object":"model","created":1764010580,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.5-opus-20251124","hugging_face_id":"","name":"Anthropic: Claude Opus 4.5","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000025","web_search":"0.01","input_cache_read":"0.0000005","input_cache_write":"0.00000625","input_cache_write_1h":"0.00001"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.5"}},{"id":"anthropic/claude-opus-4.5:batch","object":"model","created":1764010580,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.5-opus-20251124","hugging_face_id":"","name":"Anthropic: Claude Opus 4.5 (batch)","description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["file","image","text"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","web_search":"0.01","input_cache_read":"0.00000025","input_cache_write":"0.000003125","input_cache_write_1h":"0.000005"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","verbosity"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.5:batch"}},{"id":"google/gemini-3-pro-image-preview","object":"model","created":1763653797,"owned_by":"google","canonical_slug":"google/gemini-3-pro-image-preview-20251120","hugging_face_id":"","name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","context_length":65536,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000012","image":"0.000002","image_output":"0.00012","audio":"0.000002","input_audio_cache":"0.0000002","web_search":"0.014","internal_reasoning":"0.000012","input_cache_read":"0.0000002","input_cache_write":"0.000000375"},"top_provider":{"context_length":65536,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-3-pro-image-preview"}},{"id":"thenlper/gte-base","object":"model","created":1763433820,"owned_by":"thenlper","canonical_slug":"thenlper/gte-base-20251117","hugging_face_id":"thenlper/gte-base","name":"Thenlper: GTE-Base","description":"The gte-base embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, delivering efficient and effective semantic embeddings optimized for textual similarity, semantic search, and clustering applications.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/thenlper/gte-base"}},{"id":"thenlper/gte-large","object":"model","created":1763433655,"owned_by":"thenlper","canonical_slug":"thenlper/gte-large-20251117","hugging_face_id":"thenlper/gte-large","name":"Thenlper: GTE-Large","description":"The gte-large embedding model converts English sentences, paragraphs and moderate-length documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for information retrieval, semantic textual similarity, reranking and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/thenlper/gte-large"}},{"id":"intfloat/e5-large-v2","object":"model","created":1763433432,"owned_by":"intfloat","canonical_slug":"intfloat/e5-large-v2-20251117","hugging_face_id":"intfloat/e5-large-v2","name":"Intfloat: E5-Large-v2","description":"The e5-large-v2 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-accuracy semantic embeddings optimized for retrieval, semantic search, reranking, and similarity-scoring tasks.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/intfloat/e5-large-v2"}},{"id":"intfloat/e5-base-v2","object":"model","created":1763433192,"owned_by":"intfloat","canonical_slug":"intfloat/e5-base-v2-20251117","hugging_face_id":"intfloat/e5-base-v2","name":"Intfloat: E5-Base-v2","description":"The e5-base-v2 embedding model encodes English sentences and paragraphs into a 768-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, similarity scoring,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/intfloat/e5-base-v2"}},{"id":"intfloat/multilingual-e5-large","object":"model","created":1763433047,"owned_by":"intfloat","canonical_slug":"intfloat/multilingual-e5-large-20251117","hugging_face_id":"intfloat/multilingual-e5-large","name":"Intfloat: Multilingual-E5-Large","description":"The multilingual-e5-large embedding model encodes sentences, paragraphs, and documents across over 90 languages into a 1024-dimensional dense vector space, delivering robust semantic embeddings optimized for multilingual retrieval, cross-language similarity, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/intfloat/multilingual-e5-large"}},{"id":"sentence-transformers/paraphrase-minilm-l6-v2","object":"model","created":1763432454,"owned_by":"sentence-transformers","canonical_slug":"sentence-transformers/paraphrase-minilm-l6-v2-20251117","hugging_face_id":"sentence-transformers/paraphrase-MiniLM-L6-v2","name":"Sentence Transformers: paraphrase-MiniLM-L6-v2","description":"The paraphrase-MiniLM-L6-v2 embedding model converts sentences and short paragraphs into a 384-dimensional dense vector space, producing high-quality semantic embeddings optimized for paraphrase detection, semantic similarity scoring, clustering, and lightweight retrieval...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sentence-transformers/paraphrase-minilm-l6-v2"}},{"id":"sentence-transformers/all-minilm-l12-v2","object":"model","created":1763432155,"owned_by":"sentence-transformers","canonical_slug":"sentence-transformers/all-minilm-l12-v2-20251117","hugging_face_id":"sentence-transformers/all-MiniLM-L12-v2","name":"Sentence Transformers: all-MiniLM-L12-v2","description":"The all-MiniLM-L12-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, producing efficient and high-quality semantic embeddings optimized for tasks such as semantic search, clustering, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sentence-transformers/all-minilm-l12-v2"}},{"id":"baai/bge-base-en-v1.5","object":"model","created":1763431837,"owned_by":"baai","canonical_slug":"baai/bge-base-en-v1.5-20251117","hugging_face_id":"BAAI/bge-base-en-v1.5","name":"BAAI: bge-base-en-v1.5","description":"The bge-base-en-v1.5 embedding model converts English sentences and paragraphs into 768-dimensional dense vectors, delivering efficient, high-quality semantic embeddings optimized for retrieval, semantic search, and document-matching workflows. This version (v1.5) features...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/baai/bge-base-en-v1.5"}},{"id":"sentence-transformers/multi-qa-mpnet-base-dot-v1","object":"model","created":1763431339,"owned_by":"sentence-transformers","canonical_slug":"sentence-transformers/multi-qa-mpnet-base-dot-v1-20251117","hugging_face_id":"sentence-transformers/multi-qa-mpnet-base-dot-v1","name":"Sentence Transformers: multi-qa-mpnet-base-dot-v1","description":"The multi-qa-mpnet-base-dot-v1 embedding model transforms sentences and short paragraphs into a 768-dimensional dense vector space, generating high-quality semantic embeddings optimized for question-and-answer retrieval, semantic search, and similarity-scoring across diverse content.","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sentence-transformers/multi-qa-mpnet-base-dot-v1"}},{"id":"baai/bge-large-en-v1.5","object":"model","created":1763431087,"owned_by":"baai","canonical_slug":"baai/bge-large-en-v1.5-20251117","hugging_face_id":"BAAI/bge-large-en-v1.5","name":"BAAI: bge-large-en-v1.5","description":"The bge-large-en-v1.5 embedding model maps English sentences, paragraphs, and documents into a 1024-dimensional dense vector space, delivering high-fidelity semantic embeddings optimized for semantic search, document retrieval, and downstream NLP tasks...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/baai/bge-large-en-v1.5"}},{"id":"baai/bge-m3","object":"model","created":1763424372,"owned_by":"baai","canonical_slug":"baai/bge-m3-20251117","hugging_face_id":"BAAI/bge-m3","name":"BAAI: bge-m3","description":"The bge-m3 embedding model encodes sentences, paragraphs, and long documents into a 1024-dimensional dense vector space, delivering high-quality semantic embeddings optimized for multilingual retrieval, semantic search, and large-context applications.","context_length":8194,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":8194,"max_completion_tokens":7374,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/baai/bge-m3"}},{"id":"sentence-transformers/all-mpnet-base-v2","object":"model","created":1763421830,"owned_by":"sentence-transformers","canonical_slug":"sentence-transformers/all-mpnet-base-v2-20251117","hugging_face_id":"sentence-transformers/all-mpnet-base-v2","name":"Sentence Transformers: all-mpnet-base-v2","description":"The all-mpnet-base-v2 embedding model encodes sentences and short paragraphs into a 768-dimensional dense vector space, providing high-fidelity semantic embeddings well suited for tasks like information retrieval, clustering, similarity scoring, and...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sentence-transformers/all-mpnet-base-v2"}},{"id":"sentence-transformers/all-minilm-l6-v2","object":"model","created":1763421176,"owned_by":"sentence-transformers","canonical_slug":"sentence-transformers/all-minilm-l6-v2-20251117","hugging_face_id":"sentence-transformers/all-MiniLM-L6-v2","name":"Sentence Transformers: all-MiniLM-L6-v2","description":"The all-MiniLM-L6-v2 embedding model maps sentences and short paragraphs into a 384-dimensional dense vector space, enabling high-quality semantic representations that are ideal for downstream tasks such as information retrieval, clustering,...","context_length":512,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000005","completion":"0"},"top_provider":{"context_length":512,"max_completion_tokens":460,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/sentence-transformers/all-minilm-l6-v2"}},{"id":"openai/gpt-5.1","object":"model","created":1763060305,"owned_by":"openai","canonical_slug":"openai/gpt-5.1-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.1"}},{"id":"openai/gpt-5.1:batch","object":"model","created":1763060305,"owned_by":"openai","canonical_slug":"openai/gpt-5.1-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1 (batch)","description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000625","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000000625"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.1:batch"}},{"id":"openai/gpt-5.1-codex","object":"model","created":1763060298,"owned_by":"openai","canonical_slug":"openai/gpt-5.1-codex-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1-Codex","description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.00000013"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.1-codex"}},{"id":"openai/gpt-5.1-codex-mini","object":"model","created":1763057820,"owned_by":"openai","canonical_slug":"openai/gpt-5.1-codex-mini-20251113","hugging_face_id":"","name":"OpenAI: GPT-5.1-Codex-Mini","description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","context_length":400000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.00000003"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5.1-codex-mini"}},{"id":"moonshotai/kimi-k2-thinking","object":"model","created":1762440622,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k2-thinking-20251106","hugging_face_id":"moonshotai/Kimi-K2-Thinking","name":"MoonshotAI: Kimi K2 Thinking","description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000025","input_cache_read":"0.00000015"},"top_provider":{"context_length":262144,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k2-thinking"}},{"id":"amazon/nova-premier-v1","object":"model","created":1761950332,"owned_by":"amazon","canonical_slug":"amazon/nova-premier-v1","hugging_face_id":"","name":"Amazon: Nova Premier 1.0","description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","context_length":1000000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.0000125","input_cache_read":"0.000000625"},"top_provider":{"context_length":1000000,"max_completion_tokens":32000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/amazon/nova-premier-v1"}},{"id":"mistralai/mistral-embed-2312","object":"model","created":1761944622,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-embed-2312","hugging_face_id":null,"name":"Mistral: Mistral Embed 2312","description":"Mistral Embed is a specialized embedding model for text data, optimized for semantic search and RAG applications. Developed by Mistral AI in late 2023, it produces 1024-dimensional vectors that effectively...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":6553,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-embed-2312"}},{"id":"google/gemini-embedding-001","object":"model","created":1761943410,"owned_by":"google","canonical_slug":"google/gemini-embedding-001","hugging_face_id":"","name":"Google: Gemini Embedding 001","description":"gemini-embedding-001 provides a unified cutting edge experience across domains, including science, legal, finance, and coding. This embedding model has consistently held a top spot on the Massive Text Embedding Benchmark...","context_length":20000,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0","image":"0","audio":"0"},"top_provider":{"context_length":20000,"max_completion_tokens":18000,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/google/gemini-embedding-001"}},{"id":"openai/text-embedding-ada-002","object":"model","created":1761865798,"owned_by":"openai","canonical_slug":"openai/text-embedding-ada-002","hugging_face_id":"","name":"OpenAI: Text Embedding Ada 002","description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/text-embedding-ada-002"}},{"id":"openai/text-embedding-ada-002:batch","object":"model","created":1761865798,"owned_by":"openai","canonical_slug":"openai/text-embedding-ada-002","hugging_face_id":"","name":"OpenAI: Text Embedding Ada 002 (batch)","description":"text-embedding-ada-002 is OpenAI's legacy text embedding model.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/text-embedding-ada-002:batch"}},{"id":"mistralai/codestral-embed-2505","object":"model","created":1761864460,"owned_by":"mistralai","canonical_slug":"mistralai/codestral-embed-2505","hugging_face_id":"","name":"Mistral: Codestral Embed 2505","description":"Mistral Codestral Embed is specially designed for code, perfect for embedding code databases, repositories, and powering coding assistants with state-of-the-art retrieval.","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":6553,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/codestral-embed-2505"}},{"id":"openai/text-embedding-3-large","object":"model","created":1761862866,"owned_by":"openai","canonical_slug":"openai/text-embedding-3-large","hugging_face_id":"","name":"OpenAI: Text Embedding 3 Large","description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000013","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/text-embedding-3-large"}},{"id":"openai/text-embedding-3-large:batch","object":"model","created":1761862866,"owned_by":"openai","canonical_slug":"openai/text-embedding-3-large","hugging_face_id":"","name":"OpenAI: Text Embedding 3 Large (batch)","description":"text-embedding-3-large is OpenAI's most capable embedding model for both english and non-english tasks. Embeddings are a numerical representation of text that can be used to measure the relatedness between two...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000065","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/text-embedding-3-large:batch"}},{"id":"openai/text-embedding-3-small","object":"model","created":1761857455,"owned_by":"openai","canonical_slug":"openai/text-embedding-3-small","hugging_face_id":"","name":"OpenAI: Text Embedding 3 Small","description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000002","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/text-embedding-3-small"}},{"id":"openai/text-embedding-3-small:batch","object":"model","created":1761857455,"owned_by":"openai","canonical_slug":"openai/text-embedding-3-small","hugging_face_id":"","name":"OpenAI: Text Embedding 3 Small (batch)","description":"text-embedding-3-small is OpenAI's improved, more performant version of the ada embedding model. Embeddings are a numerical representation of text that can be used to measure the relatedness between two pieces...","context_length":8192,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":true},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/text-embedding-3-small:batch"}},{"id":"perplexity/sonar-pro-search","object":"model","created":1761854366,"owned_by":"perplexity","canonical_slug":"perplexity/sonar-pro-search","hugging_face_id":"","name":"Perplexity: Sonar Pro Search","description":"Exclusively available on the CLSSAI API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.018"},"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p","web_search_options"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/sonar-pro-search"}},{"id":"mistralai/voxtral-small-24b-2507","object":"model","created":1761835144,"owned_by":"mistralai","canonical_slug":"mistralai/voxtral-small-24b-2507","hugging_face_id":"mistralai/Voxtral-Small-24B-2507","name":"Mistral: Voxtral Small 24B 2507","description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","context_length":32768,"architecture":{"modality":"text+file+audio->text","input_modalities":["text","audio","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003","audio":"0.0001","input_cache_read":"0.00000001"},"top_provider":{"context_length":32768,"max_completion_tokens":26214,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.2,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/mistralai/voxtral-small-24b-2507"}},{"id":"openai/gpt-oss-safeguard-20b","object":"model","created":1761752836,"owned_by":"openai","canonical_slug":"openai/gpt-oss-safeguard-20b","hugging_face_id":"openai/gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003","input_cache_read":"0.0000000375"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-oss-safeguard-20b"}},{"id":"qwen/qwen3-embedding-8b","object":"model","created":1761680622,"owned_by":"qwen","canonical_slug":"qwen/qwen3-embedding-8b","hugging_face_id":"Qwen/Qwen3-Embedding-8B","name":"Qwen: Qwen3 Embedding 8B","description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000001","completion":"0"},"top_provider":{"context_length":32000,"max_completion_tokens":28800,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-embedding-8b"}},{"id":"qwen/qwen3-embedding-4b","object":"model","created":1761662922,"owned_by":"qwen","canonical_slug":"qwen/qwen3-embedding-4b","hugging_face_id":"Qwen/Qwen3-Embedding-4B","name":"Qwen: Qwen3 Embedding 4B","description":"The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. This series inherits the exceptional multilingual capabilities, long-text...","context_length":32768,"architecture":{"modality":"text->embeddings","input_modalities":["text"],"output_modalities":["embeddings"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000002","completion":"0"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":[],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-embedding-4b"}},{"id":"minimax/minimax-m2","object":"model","created":1761252093,"owned_by":"minimax","canonical_slug":"minimax/minimax-m2","hugging_face_id":"MiniMaxAI/MiniMax-M2","name":"MiniMax: MiniMax M2","description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000012"},"top_provider":{"context_length":196608,"max_completion_tokens":176947,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-m2"}},{"id":"qwen/qwen3-vl-32b-instruct","object":"model","created":1761231332,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-32b-instruct","hugging_face_id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.000000104","completion":"0.000000416"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-vl-32b-instruct"}},{"id":"ibm-granite/granite-4.0-h-micro","object":"model","created":1760927695,"owned_by":"ibm-granite","canonical_slug":"ibm-granite/granite-4.0-h-micro","hugging_face_id":"ibm-granite/granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","context_length":131000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000000017","completion":"0.000000112"},"top_provider":{"context_length":131000,"max_completion_tokens":117900,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/ibm-granite/granite-4.0-h-micro"}},{"id":"openai/gpt-5-image-mini","object":"model","created":1760624583,"owned_by":"openai","canonical_slug":"openai/gpt-5-image-mini","hugging_face_id":"","name":"OpenAI: GPT-5 Image Mini","description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by GPT-5 Mini, with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["file","image","text"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.000002","image_output":"0.000008","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-image-mini"}},{"id":"anthropic/claude-haiku-4.5","object":"model","created":1760547638,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.5-haiku-20251001","hugging_face_id":"","name":"Anthropic: Claude Haiku 4.5","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000001","input_cache_write":"0.00000125","input_cache_write_1h":"0.000002"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-haiku-4.5"}},{"id":"anthropic/claude-haiku-4.5:batch","object":"model","created":1760547638,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.5-haiku-20251001","hugging_face_id":"","name":"Anthropic: Claude Haiku 4.5 (batch)","description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000025","web_search":"0.01","input_cache_read":"0.00000005","input_cache_write":"0.000000625","input_cache_write_1h":"0.000001"},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-haiku-4.5:batch"}},{"id":"qwen/qwen3-vl-8b-thinking","object":"model","created":1760463746,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-8b-thinking","hugging_face_id":"Qwen/Qwen3-VL-8B-Thinking","name":"Qwen: Qwen3 VL 8B Thinking","description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000018","completion":"0.0000021"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":0.95},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-vl-8b-thinking"}},{"id":"qwen/qwen3-vl-8b-instruct","object":"model","created":1760463308,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-8b-instruct","hugging_face_id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000117","completion":"0.000000455"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-vl-8b-instruct"}},{"id":"openai/gpt-5-image","object":"model","created":1760447986,"owned_by":"openai","canonical_slug":"openai/gpt-5-image","hugging_face_id":"","name":"OpenAI: GPT-5 Image","description":"GPT-5 Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","context_length":400000,"architecture":{"modality":"text+image+file->text+image","input_modalities":["image","text","file"],"output_modalities":["image","text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00001","image_output":"0.00004","web_search":"0.01","input_cache_read":"0.00000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-image"}},{"id":"google/gemini-2.5-flash-image","object":"model","created":1759870431,"owned_by":"google","canonical_slug":"google/gemini-2.5-flash-image","hugging_face_id":"","name":"Google: Nano Banana (Gemini 2.5 Flash Image)","description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","context_length":32768,"architecture":{"modality":"text+image->text+image","input_modalities":["image","text"],"output_modalities":["image","text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","image_output":"0.00003","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":32768,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/google/gemini-2.5-flash-image"}},{"id":"qwen/qwen3-vl-30b-a3b-thinking","object":"model","created":1759794479,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000024"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-vl-30b-a3b-thinking"}},{"id":"qwen/qwen3-vl-30b-a3b-instruct","object":"model","created":1759794476,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"top_provider":{"context_length":262144,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-vl-30b-a3b-instruct"}},{"id":"openai/gpt-5-pro","object":"model","created":1759776663,"owned_by":"openai","canonical_slug":"openai/gpt-5-pro-2025-10-06","hugging_face_id":"","name":"OpenAI: GPT-5 Pro","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.00012","web_search":"0.01"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-pro"}},{"id":"openai/gpt-5-pro:batch","object":"model","created":1759776663,"owned_by":"openai","canonical_slug":"openai/gpt-5-pro-2025-10-06","hugging_face_id":"","name":"OpenAI: GPT-5 Pro (batch)","description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000075","completion":"0.00006","web_search":"0.01"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-pro:batch"}},{"id":"z-ai/glm-4.6","object":"model","created":1759235576,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.6","hugging_face_id":"zai-org/GLM-4.6","name":"Z.ai: GLM 4.6","description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","context_length":204800,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000043","completion":"0.00000175","input_cache_read":"0.00000008"},"top_provider":{"context_length":198000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-4.6"}},{"id":"anthropic/claude-sonnet-4.5","object":"model","created":1759161676,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.5","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-4.5"}},{"id":"anthropic/claude-sonnet-4.5:batch","object":"model","created":1759161676,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.5-sonnet-20250929","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4.5 (batch)","description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","context_length":1000000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000015","completion":"0.0000075","web_search":"0.01","input_cache_read":"0.00000015","input_cache_write":"0.000001875","input_cache_write_1h":"0.000003","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000003","completion":"0.00001125","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":1,"top_p":1,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-4.5:batch"}},{"id":"deepseek/deepseek-v3.2-exp","object":"model","created":1759150481,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v3.2-exp","hugging_face_id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"DeepSeek: DeepSeek V3.2 Exp","description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.00000027","completion":"0.00000041"},"top_provider":{"context_length":163840,"max_completion_tokens":147456,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-07-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v3.2-exp"}},{"id":"thedrummer/cydonia-24b-v4.1","object":"model","created":1758931878,"owned_by":"thedrummer","canonical_slug":"thedrummer/cydonia-24b-v4.1","hugging_face_id":"thedrummer/cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000005","input_cache_read":"0.00000015"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/v1/models/thedrummer/cydonia-24b-v4.1"}},{"id":"relace/relace-apply-3","object":"model","created":1758891572,"owned_by":"relace","canonical_slug":"relace/relace-apply-3","hugging_face_id":"","name":"Relace: Relace Apply 3","description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000085","completion":"0.00000125"},"top_provider":{"context_length":256000,"max_completion_tokens":128000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","seed","stop"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/relace/relace-apply-3"}},{"id":"qwen/qwen3-vl-235b-a22b-thinking","object":"model","created":1758668690,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen: Qwen3 VL 235B A22B Thinking","description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000004"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.8,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-vl-235b-a22b-thinking"}},{"id":"qwen/qwen3-vl-235b-a22b-instruct","object":"model","created":1758668687,"owned_by":"qwen","canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen: Qwen3 VL 235B A22B Instruct","description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","context_length":262144,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000021","completion":"0.0000019","input_cache_read":"0.0000001"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.7,"top_p":0.8,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-vl-235b-a22b-instruct"}},{"id":"qwen/qwen3-max","object":"model","created":1758662808,"owned_by":"qwen","canonical_slug":"qwen/qwen3-max","hugging_face_id":"","name":"Qwen: Qwen3 Max","description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000078","completion":"0.0000039","input_cache_read":"0.000000156","input_cache_write":"0.000000975","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000156","completion":"0.0000078","input_cache_read":"0.000000312","input_cache_write":"0.00000195"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.00000039","input_cache_write":"0.0000024375"}]},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":1,"top_p":1,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-max"}},{"id":"qwen/qwen3-coder-plus","object":"model","created":1758662707,"owned_by":"qwen","canonical_slug":"qwen/qwen3-coder-plus","hugging_face_id":"","name":"Qwen: Qwen3 Coder Plus","description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000065","completion":"0.00000325","input_cache_read":"0.00000013","input_cache_write":"0.0000008125","overrides":[{"min_prompt_tokens":32000,"prompt":"0.00000117","completion":"0.00000585","input_cache_read":"0.000000234","input_cache_write":"0.0000014625"},{"min_prompt_tokens":128000,"prompt":"0.00000195","completion":"0.00000975","input_cache_read":"0.00000039","input_cache_write":"0.0000024375"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-coder-plus"}},{"id":"deepseek/deepseek-v3.1-terminus","object":"model","created":1758548275,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-v3.1-terminus","hugging_face_id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.000000135"},"top_provider":{"context_length":131072,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-v3.1-terminus"}},{"id":"qwen/qwen3-coder-flash","object":"model","created":1758115536,"owned_by":"qwen","canonical_slug":"qwen/qwen3-coder-flash","hugging_face_id":"","name":"Qwen: Qwen3 Coder Flash","description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.000000195","completion":"0.000000975","input_cache_read":"0.000000039","input_cache_write":"0.00000024375","overrides":[{"min_prompt_tokens":32000,"prompt":"0.000000325","completion":"0.000001625","input_cache_read":"0.000000065","input_cache_write":"0.00000040625"},{"min_prompt_tokens":128000,"prompt":"0.00000052","completion":"0.0000026","input_cache_read":"0.000000104","input_cache_write":"0.00000065"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-coder-flash"}},{"id":"qwen/qwen3-next-80b-a3b-thinking","object":"model","created":1757612284,"owned_by":"qwen","canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Thinking","name":"Qwen: Qwen3 Next 80B A3B Thinking","description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000012"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-next-80b-a3b-thinking"}},{"id":"qwen/qwen3-next-80b-a3b-instruct","object":"model","created":1757612213,"owned_by":"qwen","canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen: Qwen3 Next 80B A3B Instruct","description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000011","input_cache_read":"0.00000007"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-09-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-next-80b-a3b-instruct"}},{"id":"qwen/qwen-plus-2025-07-28","object":"model","created":1757347599,"owned_by":"qwen","canonical_slug":"qwen/qwen-plus-2025-07-28","hugging_face_id":"","name":"Qwen: Qwen Plus 0728","description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000078","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000078","completion":"0.00000234"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen-plus-2025-07-28"}},{"id":"moonshotai/kimi-k2-0905","object":"model","created":1757021147,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k2-0905","hugging_face_id":"moonshotai/Kimi-K2-Instruct-0905","name":"MoonshotAI: Kimi K2 0905","description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000025"},"top_provider":{"context_length":262144,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k2-0905"}},{"id":"qwen/qwen3-30b-a3b-thinking-2507","object":"model","created":1756399192,"owned_by":"qwen","canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","hugging_face_id":"Qwen/Qwen3-30B-A3B-Thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000024"},"top_provider":{"context_length":81920,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-30b-a3b-thinking-2507"}},{"id":"nousresearch/hermes-4-405b","object":"model","created":1756235463,"owned_by":"nousresearch","canonical_slug":"nousresearch/hermes-4-405b","hugging_face_id":"NousResearch/Hermes-4-405B","name":"Nous: Hermes 4 405B","description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000003"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/nousresearch/hermes-4-405b"}},{"id":"deepseek/deepseek-chat-v3.1","object":"model","created":1755779628,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-chat-v3.1","hugging_face_id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek: DeepSeek V3.1","description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-v3.1"},"pricing":{"prompt":"0.00000025","completion":"0.00000095","input_cache_read":"0.00000013"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-chat-v3.1"}},{"id":"mistralai/mistral-medium-3.1","object":"model","created":1755095639,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-medium-3.1","hugging_face_id":"","name":"Mistral: Mistral Medium 3.1","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":104857,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-medium-3.1"}},{"id":"mistralai/mistral-medium-3.1:batch","object":"model","created":1755095639,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-medium-3.1","hugging_face_id":"","name":"Mistral: Mistral Medium 3.1 (batch)","description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.000001","input_cache_read":"0.00000002"},"top_provider":{"context_length":131072,"max_completion_tokens":104857,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-medium-3.1:batch"}},{"id":"z-ai/glm-4.5v","object":"model","created":1754922288,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.5v","hugging_face_id":"zai-org/GLM-4.5V","name":"Z.ai: GLM 4.5V","description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","context_length":65536,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000018","input_cache_read":"0.00000011"},"top_provider":{"context_length":65536,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-4.5v"}},{"id":"openai/gpt-5","object":"model","created":1754587413,"owned_by":"openai","canonical_slug":"openai/gpt-5-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","web_search":"0.01","input_cache_read":"0.000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5"}},{"id":"openai/gpt-5:batch","object":"model","created":1754587413,"owned_by":"openai","canonical_slug":"openai/gpt-5-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 (batch)","description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000625","completion":"0.000005","web_search":"0.01","input_cache_read":"0.0000000625"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5:batch"}},{"id":"openai/gpt-5-mini","object":"model","created":1754587407,"owned_by":"openai","canonical_slug":"openai/gpt-5-mini-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Mini","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.000002","web_search":"0.01","input_cache_read":"0.000000025"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-05-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-mini"}},{"id":"openai/gpt-5-mini:batch","object":"model","created":1754587407,"owned_by":"openai","canonical_slug":"openai/gpt-5-mini-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Mini (batch)","description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000125","completion":"0.000001","web_search":"0.01","input_cache_read":"0.0000000125"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-05-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-mini:batch"}},{"id":"openai/gpt-5-nano","object":"model","created":1754587402,"owned_by":"openai","canonical_slug":"openai/gpt-5-nano-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Nano","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000005"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-05-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-nano"}},{"id":"openai/gpt-5-nano:batch","object":"model","created":1754587402,"owned_by":"openai","canonical_slug":"openai/gpt-5-nano-2025-08-07","hugging_face_id":"","name":"OpenAI: GPT-5 Nano (batch)","description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","context_length":400000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000025","completion":"0.0000002","web_search":"0.01","input_cache_read":"0.0000000025"},"top_provider":{"context_length":400000,"max_completion_tokens":128000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools","verbosity"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-05-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-5-nano:batch"}},{"id":"openai/gpt-oss-120b","object":"model","created":1754414231,"owned_by":"openai","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000037","completion":"0.00000017"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_a","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-oss-120b"}},{"id":"openai/gpt-oss-120b:batch","object":"model","created":1754414231,"owned_by":"openai","canonical_slug":"openai/gpt-oss-120b","hugging_face_id":"openai/gpt-oss-120b","name":"OpenAI: gpt-oss-120b (batch)","description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000000296","completion":"0.000000136"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-oss-120b:batch"}},{"id":"openai/gpt-oss-20b","object":"model","created":1754414229,"owned_by":"openai","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000018","completion":"0.00000009","input_cache_read":"0.000000009"},"top_provider":{"context_length":131072,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-oss-20b"}},{"id":"openai/gpt-oss-20b:batch","object":"model","created":1754414229,"owned_by":"openai","canonical_slug":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"OpenAI: gpt-oss-20b (batch)","description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000024","completion":"0.000000112"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-oss-20b:batch"}},{"id":"anthropic/claude-opus-4.1","object":"model","created":1754411591,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.1-opus-20250805","hugging_face_id":"","name":"Anthropic: Claude Opus 4.1","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.000075","web_search":"0.01","input_cache_read":"0.0000015","input_cache_write":"0.00001875","input_cache_write_1h":"0.00003"},"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.1"}},{"id":"anthropic/claude-opus-4.1:batch","object":"model","created":1754411591,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4.1-opus-20250805","hugging_face_id":"","name":"Anthropic: Claude Opus 4.1 (batch)","description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.0000075","completion":"0.0000375","web_search":"0.01","input_cache_read":"0.00000075","input_cache_write":"0.000009375","input_cache_write_1h":"0.000015"},"top_provider":{"context_length":200000,"max_completion_tokens":32000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-opus-4.1:batch"}},{"id":"mistralai/codestral-2508","object":"model","created":1754079630,"owned_by":"mistralai","canonical_slug":"mistralai/codestral-2508","hugging_face_id":"","name":"Mistral: Codestral 2508","description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000009","input_cache_read":"0.00000003"},"top_provider":{"context_length":256000,"max_completion_tokens":204800,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/codestral-2508"}},{"id":"mistralai/codestral-2508:batch","object":"model","created":1754079630,"owned_by":"mistralai","canonical_slug":"mistralai/codestral-2508","hugging_face_id":"","name":"Mistral: Codestral 2508 (batch)","description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","context_length":256000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000045","input_cache_read":"0.000000015"},"top_provider":{"context_length":256000,"max_completion_tokens":204800,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/codestral-2508:batch"}},{"id":"qwen/qwen3-coder-30b-a3b-instruct","object":"model","created":1753972379,"owned_by":"qwen","canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","hugging_face_id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen: Qwen3 Coder 30B A3B Instruct","description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.00000028"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-coder-30b-a3b-instruct"}},{"id":"qwen/qwen3-30b-a3b-instruct-2507","object":"model","created":1753806965,"owned_by":"qwen","canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","hugging_face_id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-30b-a3b-instruct-2507"}},{"id":"z-ai/glm-4.5","object":"model","created":1753471347,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.5","hugging_face_id":"zai-org/GLM-4.5","name":"Z.ai: GLM 4.5","description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000006","completion":"0.0000022","input_cache_read":"0.00000011"},"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":"2026-12-31","links":{"details":"/v1/models/z-ai/glm-4.5"}},{"id":"z-ai/glm-4.5-air","object":"model","created":1753471258,"owned_by":"z-ai","canonical_slug":"z-ai/glm-4.5-air","hugging_face_id":"zai-org/GLM-4.5-Air","name":"Z.ai: GLM 4.5 Air","description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000013","completion":"0.00000085","input_cache_read":"0.000000025"},"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.75,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/v1/models/z-ai/glm-4.5-air"}},{"id":"qwen/qwen3-235b-a22b-thinking-2507","object":"model","created":1753449557,"owned_by":"qwen","canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","hugging_face_id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000023","completion":"0.0000023"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-235b-a22b-thinking-2507"}},{"id":"qwen/qwen3-coder","object":"model","created":1753230546,"owned_by":"qwen","canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","hugging_face_id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen: Qwen3 Coder 480B A35B","description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.000001","input_cache_read":"0.0000001"},"top_provider":{"context_length":262144,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-coder"}},{"id":"bytedance/ui-tars-1.5-7b","object":"model","created":1753205056,"owned_by":"bytedance","canonical_slug":"bytedance/ui-tars-1.5-7b","hugging_face_id":"ByteDance-Seed/UI-TARS-1.5-7B","name":"ByteDance: UI-TARS 7B ","description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000002","input_cache_read":"0.0000001"},"top_provider":{"context_length":128000,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/bytedance/ui-tars-1.5-7b"}},{"id":"google/gemini-2.5-flash-lite","object":"model","created":1753200276,"owned_by":"google","canonical_slug":"google/gemini-2.5-flash-lite","hugging_face_id":"","name":"Google: Gemini 2.5 Flash Lite","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","image":"0.0000001","audio":"0.0000003","input_audio_cache":"0.00000003","web_search":"0.014","internal_reasoning":"0.0000004","input_cache_read":"0.00000001","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":"2026-10-20","links":{"details":"/v1/models/google/gemini-2.5-flash-lite"}},{"id":"google/gemini-2.5-flash-lite:batch","object":"model","created":1753200276,"owned_by":"google","canonical_slug":"google/gemini-2.5-flash-lite","hugging_face_id":"","name":"Google: Gemini 2.5 Flash Lite (batch)","description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000002","image":"0.00000005","audio":"0.00000015","input_audio_cache":"0.00000003","web_search":"0.014","internal_reasoning":"0.0000002","input_cache_read":"0.00000001"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/google/gemini-2.5-flash-lite:batch"}},{"id":"qwen/qwen3-235b-a22b-2507","object":"model","created":1753119555,"owned_by":"qwen","canonical_slug":"qwen/qwen3-235b-a22b-07-25","hugging_face_id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":null},"pricing":{"prompt":"0.0000000875","completion":"0.00000035","input_cache_read":"0.0000000175"},"top_provider":{"context_length":262144,"max_completion_tokens":235929,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-235b-a22b-2507"}},{"id":"moonshotai/kimi-k2","object":"model","created":1752263252,"owned_by":"moonshotai","canonical_slug":"moonshotai/kimi-k2","hugging_face_id":"moonshotai/Kimi-K2-Instruct","name":"MoonshotAI: Kimi K2 0711","description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000057","completion":"0.0000023"},"top_provider":{"context_length":131072,"max_completion_tokens":98304,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-12-31","expiration_date":null,"links":{"details":"/v1/models/moonshotai/kimi-k2"}},{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","object":"model","created":1752094966,"owned_by":"cognitivecomputations","canonical_slug":"venice/uncensored","hugging_face_id":"cognitivecomputations/Dolphin-Mistral-24B-Venice-Edition","name":"Venice: Uncensored","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000009"},"top_provider":{"context_length":128000,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/v1/models/cognitivecomputations/dolphin-mistral-24b-venice-edition"}},{"id":"tencent/hunyuan-a13b-instruct","object":"model","created":1751987664,"owned_by":"tencent","canonical_slug":"tencent/hunyuan-a13b-instruct","hugging_face_id":"tencent/Hunyuan-A13B-Instruct","name":"Tencent: Hunyuan A13B Instruct","description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000014","completion":"0.00000057"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/tencent/hunyuan-a13b-instruct"}},{"id":"morph/morph-v3-large","object":"model","created":1751910858,"owned_by":"morph","canonical_slug":"morph/morph-v3-large","hugging_face_id":"","name":"Morph: Morph V3 Large","description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","context_length":262144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000009","completion":"0.0000019"},"top_provider":{"context_length":262144,"max_completion_tokens":131072,"is_moderated":false},"per_request_limits":null,"supported_parameters":["logprobs","max_tokens","response_format","stop","structured_outputs","temperature","top_logprobs"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/morph/morph-v3-large"}},{"id":"morph/morph-v3-fast","object":"model","created":1751910002,"owned_by":"morph","canonical_slug":"morph/morph-v3-fast","hugging_face_id":"","name":"Morph: Morph V3 Fast","description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","context_length":81920,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000012"},"top_provider":{"context_length":81920,"max_completion_tokens":38000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/morph/morph-v3-fast"}},{"id":"baidu/ernie-4.5-vl-424b-a47b","object":"model","created":1751300903,"owned_by":"baidu","canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","hugging_face_id":"baidu/ERNIE-4.5-VL-424B-A47B-PT","name":"Baidu: ERNIE 4.5 VL 424B A47B ","description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","context_length":123000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000042","completion":"0.00000125"},"top_provider":{"context_length":123000,"max_completion_tokens":16000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":"2026-10-08","links":{"details":"/v1/models/baidu/ernie-4.5-vl-424b-a47b"}},{"id":"mistralai/mistral-small-3.2-24b-instruct","object":"model","created":1750443016,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","hugging_face_id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral: Mistral Small 3.2 24B","description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","context_length":256000,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000009375","completion":"0.00000025"},"top_provider":{"context_length":256000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-small-3.2-24b-instruct"}},{"id":"minimax/minimax-m1","object":"model","created":1750200414,"owned_by":"minimax","canonical_slug":"minimax/minimax-m1","hugging_face_id":"","name":"MiniMax: MiniMax M1","description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000022"},"top_provider":{"context_length":1000000,"max_completion_tokens":40000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-m1"}},{"id":"google/gemini-2.5-flash","object":"model","created":1750172488,"owned_by":"google","canonical_slug":"google/gemini-2.5-flash","hugging_face_id":"","name":"Google: Gemini 2.5 Flash","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.0000003","completion":"0.0000025","image":"0.0000003","audio":"0.000001","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.0000025","input_cache_read":"0.00000003","input_cache_write":"0.0000000833333333333333"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":"2026-10-20","links":{"details":"/v1/models/google/gemini-2.5-flash"}},{"id":"google/gemini-2.5-flash:batch","object":"model","created":1750172488,"owned_by":"google","canonical_slug":"google/gemini-2.5-flash","hugging_face_id":"","name":"Google: Gemini 2.5 Flash (batch)","description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["file","image","text","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.00000125","image":"0.00000015","audio":"0.0000005","input_audio_cache":"0.0000001","web_search":"0.014","internal_reasoning":"0.00000125","input_cache_read":"0.00000003"},"top_provider":{"context_length":1048576,"max_completion_tokens":65535,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":"2026-10-20","links":{"details":"/v1/models/google/gemini-2.5-flash:batch"}},{"id":"google/gemini-2.5-pro","object":"model","created":1750169544,"owned_by":"google","canonical_slug":"google/gemini-2.5-pro","hugging_face_id":"","name":"Google: Gemini 2.5 Pro","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","image":"0.00000125","audio":"0.00000125","input_audio_cache":"0.000000125","web_search":"0.014","internal_reasoning":"0.00001","input_cache_read":"0.000000125","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000015","audio":"0.0000025","input_audio_cache":"0.00000025","input_cache_read":"0.00000025"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":"2026-10-20","links":{"details":"/v1/models/google/gemini-2.5-pro"}},{"id":"google/gemini-2.5-pro:batch","object":"model","created":1750169544,"owned_by":"google","canonical_slug":"google/gemini-2.5-pro","hugging_face_id":"","name":"Google: Gemini 2.5 Pro (batch)","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio+video->text","input_modalities":["text","image","file","audio","video"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.000000625","completion":"0.000005","image":"0.000000625","audio":"0.000000625","input_audio_cache":"0.000000125","web_search":"0.014","internal_reasoning":"0.000005","input_cache_read":"0.000000125","overrides":[{"min_prompt_tokens":200000,"prompt":"0.00000125","completion":"0.0000075","audio":"0.00000125","input_audio_cache":"0.00000025","input_cache_read":"0.00000025"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":"2026-10-20","links":{"details":"/v1/models/google/gemini-2.5-pro:batch"}},{"id":"openai/o3-pro","object":"model","created":1749598352,"owned_by":"openai","canonical_slug":"openai/o3-pro-2025-06-10","hugging_face_id":"","name":"OpenAI: o3 Pro","description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","file","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00002","completion":"0.00008","web_search":"0.01"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/o3-pro"}},{"id":"google/gemini-2.5-pro-preview","object":"model","created":1749137257,"owned_by":"google","canonical_slug":"google/gemini-2.5-pro-preview-06-05","hugging_face_id":"","name":"Google: Gemini 2.5 Pro Preview 06-05","description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","context_length":1048576,"architecture":{"modality":"text+image+file+audio->text","input_modalities":["file","image","text","audio"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.00001","image":"0.00000125","audio":"0.00000125","input_audio_cache":"0.000000125","web_search":"0.014","internal_reasoning":"0.00001","input_cache_read":"0.000000125","input_cache_write":"0.000000375","overrides":[{"min_prompt_tokens":200000,"prompt":"0.0000025","completion":"0.000015","audio":"0.0000025","input_audio_cache":"0.00000025","input_cache_read":"0.00000025"}]},"top_provider":{"context_length":1048576,"max_completion_tokens":65536,"is_moderated":false},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/google/gemini-2.5-pro-preview"}},{"id":"deepseek/deepseek-r1-0528","object":"model","created":1748455170,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-r1-0528","hugging_face_id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek: R1 0528","description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.0000005","completion":"0.00000215","input_cache_read":"0.00000035"},"top_provider":{"context_length":163840,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-r1-0528"}},{"id":"anthropic/claude-sonnet-4","object":"model","created":1747930371,"owned_by":"anthropic","canonical_slug":"anthropic/claude-4-sonnet-20250522","hugging_face_id":"","name":"Anthropic: Claude Sonnet 4","description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"Claude","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.01","input_cache_read":"0.0000003","input_cache_write":"0.00000375","input_cache_write_1h":"0.000006","overrides":[{"min_prompt_tokens":200000,"prompt":"0.000006","completion":"0.0000225","input_cache_read":"0.0000006","input_cache_write":"0.0000075","input_cache_write_1h":"0.000012"}]},"top_provider":{"context_length":200000,"max_completion_tokens":64000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/anthropic/claude-sonnet-4"}},{"id":"mistralai/mistral-medium-3","object":"model","created":1746627341,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-medium-3","hugging_face_id":"","name":"Mistral: Mistral Medium 3","description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","context_length":131072,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.000002","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":104857,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-medium-3"}},{"id":"meta-llama/llama-guard-4-12b","object":"model","created":1745975193,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-guard-4-12b","hugging_face_id":"meta-llama/Llama-Guard-4-12B","name":"Meta: Llama Guard 4 12B","description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","context_length":163840,"architecture":{"modality":"text+image->text","input_modalities":["image","text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000018","completion":"0.00000018"},"top_provider":{"context_length":163840,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-guard-4-12b"}},{"id":"qwen/qwen3-30b-a3b","object":"model","created":1745878604,"owned_by":"qwen","canonical_slug":"qwen/qwen3-30b-a3b-04-28","hugging_face_id":"Qwen/Qwen3-30B-A3B","name":"Qwen: Qwen3 30B A3B","description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000012","completion":"0.0000005"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-30b-a3b"}},{"id":"qwen/qwen3-8b","object":"model","created":1745876632,"owned_by":"qwen","canonical_slug":"qwen/qwen3-8b-04-28","hugging_face_id":"Qwen/Qwen3-8B","name":"Qwen: Qwen3 8B","description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.000000117","completion":"0.000000455"},"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":0.6,"top_p":0.95,"top_k":20,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-8b"}},{"id":"qwen/qwen3-14b","object":"model","created":1745876478,"owned_by":"qwen","canonical_slug":"qwen/qwen3-14b-04-28","hugging_face_id":"Qwen/Qwen3-14B","name":"Qwen: Qwen3 14B","description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000012","completion":"0.00000024"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-14b"}},{"id":"qwen/qwen3-32b","object":"model","created":1745875945,"owned_by":"qwen","canonical_slug":"qwen/qwen3-32b-04-28","hugging_face_id":"Qwen/Qwen3-32B","name":"Qwen: Qwen3 32B","description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.00000008","completion":"0.00000028"},"top_provider":{"context_length":40960,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen3-32b"}},{"id":"qwen/qwen3-235b-a22b","object":"model","created":1745875757,"owned_by":"qwen","canonical_slug":"qwen/qwen3-235b-a22b-04-28","hugging_face_id":"Qwen/Qwen3-235B-A22B","name":"Qwen: Qwen3 235B A22B","description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen3","instruct_type":"qwen3"},"pricing":{"prompt":"0.000000455","completion":"0.00000182"},"top_provider":{"context_length":131072,"max_completion_tokens":8192,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":"2026-10-09","links":{"details":"/v1/models/qwen/qwen3-235b-a22b"}},{"id":"openai/o4-mini-high","object":"model","created":1744824212,"owned_by":"openai","canonical_slug":"openai/o4-mini-high-2025-04-16","hugging_face_id":"","name":"OpenAI: o4 Mini High","description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/o4-mini-high"}},{"id":"openai/o3","object":"model","created":1744823457,"owned_by":"openai","canonical_slug":"openai/o3-2025-04-16","hugging_face_id":"","name":"OpenAI: o3","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/o3"}},{"id":"openai/o3:batch","object":"model","created":1744823457,"owned_by":"openai","canonical_slug":"openai/o3-2025-04-16","hugging_face_id":"","name":"OpenAI: o3 (batch)","description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000004","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/o3:batch"}},{"id":"openai/o4-mini","object":"model","created":1744820942,"owned_by":"openai","canonical_slug":"openai/o4-mini-2025-04-16","hugging_face_id":"","name":"OpenAI: o4 Mini","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.000000275"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/o4-mini"}},{"id":"openai/o4-mini:batch","object":"model","created":1744820942,"owned_by":"openai","canonical_slug":"openai/o4-mini-2025-04-16","hugging_face_id":"","name":"OpenAI: o4 Mini (batch)","description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000022","web_search":"0.01","input_cache_read":"0.0000001375"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/o4-mini:batch"}},{"id":"openai/gpt-4.1","object":"model","created":1744651385,"owned_by":"openai","canonical_slug":"openai/gpt-4.1-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.01","input_cache_read":"0.0000005"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4.1"}},{"id":"openai/gpt-4.1:batch","object":"model","created":1744651385,"owned_by":"openai","canonical_slug":"openai/gpt-4.1-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 (batch)","description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000004","web_search":"0.01","input_cache_read":"0.00000025"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4.1:batch"}},{"id":"openai/gpt-4.1-mini","object":"model","created":1744651381,"owned_by":"openai","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 Mini","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000004","completion":"0.0000016","web_search":"0.01","input_cache_read":"0.0000001"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4.1-mini"}},{"id":"openai/gpt-4.1-mini:batch","object":"model","created":1744651381,"owned_by":"openai","canonical_slug":"openai/gpt-4.1-mini-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 Mini (batch)","description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000008","web_search":"0.01","input_cache_read":"0.00000005"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4.1-mini:batch"}},{"id":"openai/gpt-4.1-nano","object":"model","created":1744651369,"owned_by":"openai","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 Nano","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000004","web_search":"0.01","input_cache_read":"0.000000025"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4.1-nano"}},{"id":"openai/gpt-4.1-nano:batch","object":"model","created":1744651369,"owned_by":"openai","canonical_slug":"openai/gpt-4.1-nano-2025-04-14","hugging_face_id":"","name":"OpenAI: GPT-4.1 Nano (batch)","description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","context_length":1047576,"architecture":{"modality":"text+image+file->text","input_modalities":["image","text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.0000002","web_search":"0.01","input_cache_read":"0.0000000125"},"top_provider":{"context_length":1047576,"max_completion_tokens":32768,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4.1-nano:batch"}},{"id":"meta-llama/llama-4-maverick","object":"model","created":1743881822,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","hugging_face_id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","name":"Meta: Llama 4 Maverick","description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","context_length":1048576,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":"0.0000001875","completion":"0.0000006525"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-4-maverick"}},{"id":"meta-llama/llama-4-scout","object":"model","created":1743881519,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","hugging_face_id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","name":"Meta: Llama 4 Scout","description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","context_length":1310720,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Llama4","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000003"},"top_provider":{"context_length":327680,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-4-scout"}},{"id":"deepseek/deepseek-chat-v3-0324","object":"model","created":1742824755,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-chat-v3-0324","hugging_face_id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek: DeepSeek V3 0324","description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.00000029","completion":"0.00000114","input_cache_read":"0.00000011"},"top_provider":{"context_length":128000,"max_completion_tokens":115200,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-chat-v3-0324"}},{"id":"openai/o1-pro","object":"model","created":1742423211,"owned_by":"openai","canonical_slug":"openai/o1-pro","hugging_face_id":"","name":"OpenAI: o1-pro","description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00015","completion":"0.0006","web_search":"0.01"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/o1-pro"}},{"id":"mistralai/mistral-small-3.1-24b-instruct","object":"model","created":1742238937,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","hugging_face_id":"mistralai/Mistral-Small-3.1-24B-Instruct-2503","name":"Mistral: Mistral Small 3.1 24B","description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000000351","completion":"0.000000555"},"top_provider":{"context_length":128000,"max_completion_tokens":102400,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-small-3.1-24b-instruct"}},{"id":"google/gemma-3-4b-it","object":"model","created":1741905510,"owned_by":"google","canonical_slug":"google/gemma-3-4b-it","hugging_face_id":"google/gemma-3-4b-it","name":"Google: Gemma 3 4B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000005","completion":"0.0000001"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/google/gemma-3-4b-it"}},{"id":"google/gemma-3-12b-it","object":"model","created":1741902625,"owned_by":"google","canonical_slug":"google/gemma-3-12b-it","hugging_face_id":"google/gemma-3-12b-it","name":"Google: Gemma 3 12B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000005","completion":"0.00000015"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/google/gemma-3-12b-it"}},{"id":"cohere/command-a","object":"model","created":1741894342,"owned_by":"cohere","canonical_slug":"cohere/command-a-03-2025","hugging_face_id":"CohereForAI/c4ai-command-a-03-2025","name":"Cohere: Command A","description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","context_length":256000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":256000,"max_completion_tokens":8192,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/cohere/command-a"}},{"id":"rekaai/reka-flash-3","object":"model","created":1741812813,"owned_by":"rekaai","canonical_slug":"rekaai/reka-flash-3","hugging_face_id":"RekaAI/reka-flash-3","name":"Reka Flash 3","description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","context_length":65536,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"top_provider":{"context_length":65536,"max_completion_tokens":58982,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2025-01-31","expiration_date":null,"links":{"details":"/v1/models/rekaai/reka-flash-3"}},{"id":"google/gemma-3-27b-it","object":"model","created":1741756359,"owned_by":"google","canonical_slug":"google/gemma-3-27b-it","hugging_face_id":"google/gemma-3-27b-it","name":"Google: Gemma 3 27B","description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","context_length":131072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000008","completion":"0.00000045","input_cache_read":"0.00000004"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/google/gemma-3-27b-it"}},{"id":"thedrummer/skyfall-36b-v2","object":"model","created":1741636566,"owned_by":"thedrummer","canonical_slug":"thedrummer/skyfall-36b-v2","hugging_face_id":"TheDrummer/Skyfall-36B-v2","name":"TheDrummer: Skyfall 36B V2","description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000008","input_cache_read":"0.00000025"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/thedrummer/skyfall-36b-v2"}},{"id":"perplexity/sonar-reasoning-pro","object":"model","created":1741313308,"owned_by":"perplexity","canonical_slug":"perplexity/sonar-reasoning-pro","hugging_face_id":"","name":"Perplexity: Sonar Reasoning Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005"},"top_provider":{"context_length":128000,"max_completion_tokens":115200,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/sonar-reasoning-pro"}},{"id":"perplexity/sonar-pro","object":"model","created":1741312423,"owned_by":"perplexity","canonical_slug":"perplexity/sonar-pro","hugging_face_id":"","name":"Perplexity: Sonar Pro","description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","context_length":200000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000015","web_search":"0.005"},"top_provider":{"context_length":200000,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/sonar-pro"}},{"id":"perplexity/sonar-deep-research","object":"model","created":1741311246,"owned_by":"perplexity","canonical_slug":"perplexity/sonar-deep-research","hugging_face_id":"","name":"Perplexity: Sonar Deep Research","description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.000002","completion":"0.000008","web_search":"0.005","internal_reasoning":"0.000003"},"top_provider":{"context_length":128000,"max_completion_tokens":115200,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/sonar-deep-research"}},{"id":"mistralai/mistral-saba","object":"model","created":1739803239,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-saba-2502","hugging_face_id":"","name":"Mistral: Saba","description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","context_length":32768,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000006","input_cache_read":"0.00000002"},"top_provider":{"context_length":32768,"max_completion_tokens":26214,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-09-30","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-saba"}},{"id":"openai/o3-mini-high","object":"model","created":1739372611,"owned_by":"openai","canonical_slug":"openai/o3-mini-high-2025-01-31","hugging_face_id":"","name":"OpenAI: o3 Mini High","description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.00000055"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","reasoning_effort","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/o3-mini-high"}},{"id":"aion-labs/aion-rp-llama-3.1-8b","object":"model","created":1738696718,"owned_by":"aion-labs","canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","hugging_face_id":"","name":"AionLabs: Aion-RP 1.0 (8B)","description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000016"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/aion-labs/aion-rp-llama-3.1-8b"}},{"id":"qwen/qwen2.5-vl-72b-instruct","object":"model","created":1738410311,"owned_by":"qwen","canonical_slug":"qwen/qwen2.5-vl-72b-instruct","hugging_face_id":"Qwen/Qwen2.5-VL-72B-Instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.000001","input_cache_read":"0.0000004"},"top_provider":{"context_length":128000,"max_completion_tokens":115200,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen2.5-vl-72b-instruct"}},{"id":"qwen/qwen-plus","object":"model","created":1738409840,"owned_by":"qwen","canonical_slug":"qwen/qwen-plus-2025-01-25","hugging_face_id":"","name":"Qwen: Qwen-Plus","description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","context_length":1000000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":null},"pricing":{"prompt":"0.00000026","completion":"0.00000078","input_cache_read":"0.000000052","input_cache_write":"0.000000325","overrides":[{"min_prompt_tokens":256000,"prompt":"0.00000078","completion":"0.00000234","input_cache_read":"0.000000156","input_cache_write":"0.000000975"}]},"top_provider":{"context_length":1000000,"max_completion_tokens":32768,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2025-03-31","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-plus"}},{"id":"openai/o3-mini","object":"model","created":1738351721,"owned_by":"openai","canonical_slug":"openai/o3-mini-2025-01-31","hugging_face_id":"","name":"OpenAI: o3 Mini","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000011","completion":"0.0000044","web_search":"0.01","input_cache_read":"0.00000055"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/o3-mini"}},{"id":"openai/o3-mini:batch","object":"model","created":1738351721,"owned_by":"openai","canonical_slug":"openai/o3-mini-2025-01-31","hugging_face_id":"","name":"OpenAI: o3 Mini (batch)","description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","context_length":200000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000055","completion":"0.0000022","web_search":"0.01","input_cache_read":"0.000000275"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/o3-mini:batch"}},{"id":"mistralai/mistral-small-24b-instruct-2501","object":"model","created":1738255409,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-small-24b-instruct-2501","hugging_face_id":"mistralai/Mistral-Small-24B-Instruct-2501","name":"Mistral: Mistral Small 3","description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.00000005","completion":"0.00000008"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{"temperature":0.3,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-small-24b-instruct-2501"}},{"id":"perplexity/sonar","object":"model","created":1738013808,"owned_by":"perplexity","canonical_slug":"perplexity/sonar","hugging_face_id":"","name":"Perplexity: Sonar","description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","context_length":127072,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000001","web_search":"0.005"},"top_provider":{"context_length":127072,"max_completion_tokens":114364,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":null,"expiration_date":null,"links":{"details":"/v1/models/perplexity/sonar"}},{"id":"deepseek/deepseek-r1","object":"model","created":1737381095,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-r1","hugging_face_id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek: R1","description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","context_length":64000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":"deepseek-r1"},"pricing":{"prompt":"0.0000007","completion":"0.0000025"},"top_provider":{"context_length":64000,"max_completion_tokens":16000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-r1"}},{"id":"minimax/minimax-01","object":"model","created":1736915462,"owned_by":"minimax","canonical_slug":"minimax/minimax-01","hugging_face_id":"MiniMaxAI/MiniMax-Text-01","name":"MiniMax: MiniMax-01","description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","context_length":1000192,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.0000002","completion":"0.0000011"},"top_provider":{"context_length":1000192,"max_completion_tokens":900172,"is_moderated":false},"per_request_limits":null,"supported_parameters":["max_tokens","temperature","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/v1/models/minimax/minimax-01"}},{"id":"microsoft/phi-4","object":"model","created":1736489872,"owned_by":"microsoft","canonical_slug":"microsoft/phi-4","hugging_face_id":"microsoft/phi-4","name":"Microsoft: Phi 4","description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","context_length":16384,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Other","instruct_type":null},"pricing":{"prompt":"0.00000007","completion":"0.00000014"},"top_provider":{"context_length":16384,"max_completion_tokens":14745,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/microsoft/phi-4"}},{"id":"deepseek/deepseek-chat","object":"model","created":1735241320,"owned_by":"deepseek","canonical_slug":"deepseek/deepseek-chat-v3","hugging_face_id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek: DeepSeek V3","description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","context_length":163840,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"DeepSeek","instruct_type":null},"pricing":{"prompt":"0.0000002574","completion":"0.0000010287"},"top_provider":{"context_length":128000,"max_completion_tokens":16000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-07-31","expiration_date":null,"links":{"details":"/v1/models/deepseek/deepseek-chat"}},{"id":"sao10k/l3.3-euryale-70b","object":"model","created":1734535928,"owned_by":"sao10k","canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","hugging_face_id":"Sao10K/L3.3-70B-Euryale-v2.3","name":"Sao10K: Llama 3.3 Euryale 70B","description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000065","completion":"0.00000075"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/sao10k/l3.3-euryale-70b"}},{"id":"openai/o1","object":"model","created":1734459999,"owned_by":"openai","canonical_slug":"openai/o1-2024-12-17","hugging_face_id":"","name":"OpenAI: o1","description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","context_length":200000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000015","completion":"0.00006","web_search":"0.01","input_cache_read":"0.0000075"},"top_provider":{"context_length":200000,"max_completion_tokens":100000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"default_parameters":{"temperature":null,"top_p":null,"top_k":null,"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/o1"}},{"id":"cohere/command-r7b-12-2024","object":"model","created":1734158152,"owned_by":"cohere","canonical_slug":"cohere/command-r7b-12-2024","hugging_face_id":"","name":"Cohere: Command R7B (12-2024)","description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.0000000375","completion":"0.00000015"},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-08-31","expiration_date":null,"links":{"details":"/v1/models/cohere/command-r7b-12-2024"}},{"id":"meta-llama/llama-3.3-70b-instruct","object":"model","created":1733506137,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-3.3-70b-instruct","hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","name":"Meta: Llama 3.3 70B Instruct","description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.0000001","completion":"0.00000032"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-3.3-70b-instruct"}},{"id":"amazon/nova-lite-v1","object":"model","created":1733437363,"owned_by":"amazon","canonical_slug":"amazon/nova-lite-v1","hugging_face_id":"","name":"Amazon: Nova Lite 1.0","description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.00000006","completion":"0.00000024"},"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/v1/models/amazon/nova-lite-v1"}},{"id":"amazon/nova-micro-v1","object":"model","created":1733437237,"owned_by":"amazon","canonical_slug":"amazon/nova-micro-v1","hugging_face_id":"","name":"Amazon: Nova Micro 1.0","description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.000000035","completion":"0.00000014"},"top_provider":{"context_length":128000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/v1/models/amazon/nova-micro-v1"}},{"id":"amazon/nova-pro-v1","object":"model","created":1733436303,"owned_by":"amazon","canonical_slug":"amazon/nova-pro-v1","hugging_face_id":"","name":"Amazon: Nova Pro 1.0","description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","context_length":300000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"Nova","instruct_type":null},"pricing":{"prompt":"0.0000008","completion":"0.0000032"},"top_provider":{"context_length":300000,"max_completion_tokens":5120,"is_moderated":true},"per_request_limits":null,"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-10-31","expiration_date":null,"links":{"details":"/v1/models/amazon/nova-pro-v1"}},{"id":"openai/gpt-4o-2024-11-20","object":"model","created":1732127594,"owned_by":"openai","canonical_slug":"openai/gpt-4o-2024-11-20","hugging_face_id":"","name":"OpenAI: GPT-4o (2024-11-20)","description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-2024-11-20"}},{"id":"mistralai/mistral-large-2407","object":"model","created":1731978415,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-large-2407","hugging_face_id":"","name":"Mistral Large 2407","description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":131072,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":131072,"max_completion_tokens":104857,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-large-2407"}},{"id":"qwen/qwen-2.5-coder-32b-instruct","object":"model","created":1731368400,"owned_by":"qwen","canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","hugging_face_id":"Qwen/Qwen2.5-Coder-32B-Instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000066","completion":"0.000001"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-2.5-coder-32b-instruct"}},{"id":"thedrummer/unslopnemo-12b","object":"model","created":1731103448,"owned_by":"thedrummer","canonical_slug":"thedrummer/unslopnemo-12b","hugging_face_id":"TheDrummer/UnslopNemo-12B-v4.1","name":"TheDrummer: UnslopNemo 12B","description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","context_length":1024000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"top_provider":{"context_length":1024000,"max_completion_tokens":819200,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/v1/models/thedrummer/unslopnemo-12b"}},{"id":"anthracite-org/magnum-v4-72b","object":"model","created":1729555200,"owned_by":"anthracite-org","canonical_slug":"anthracite-org/magnum-v4-72b","hugging_face_id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet() and Opus().\n\nThe model is fine-tuned on top of Qwen2.5 72B.","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.0000025","completion":"0.000005"},"top_provider":{"context_length":32768,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/anthracite-org/magnum-v4-72b"}},{"id":"qwen/qwen-2.5-7b-instruct","object":"model","created":1729036800,"owned_by":"qwen","canonical_slug":"qwen/qwen-2.5-7b-instruct","hugging_face_id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen: Qwen2.5 7B Instruct","description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.0000001","completion":"0.0000002"},"top_provider":{"context_length":32768,"max_completion_tokens":29491,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{"temperature":null,"top_p":null,"frequency_penalty":null},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-2.5-7b-instruct"}},{"id":"meta-llama/llama-3.2-1b-instruct","object":"model","created":1727222400,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-3.2-1b-instruct","hugging_face_id":"meta-llama/Llama-3.2-1B-Instruct","name":"Meta: Llama 3.2 1B Instruct","description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","context_length":60000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.000000027","completion":"0.000000201"},"top_provider":{"context_length":60000,"max_completion_tokens":54000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-3.2-1b-instruct"}},{"id":"meta-llama/llama-3.2-3b-instruct","object":"model","created":1727222400,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-3.2-3b-instruct","hugging_face_id":"meta-llama/Llama-3.2-3B-Instruct","name":"Meta: Llama 3.2 3B Instruct","description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000005","completion":"0.00000033"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-3.2-3b-instruct"}},{"id":"qwen/qwen-2.5-72b-instruct","object":"model","created":1726704000,"owned_by":"qwen","canonical_slug":"qwen/qwen-2.5-72b-instruct","hugging_face_id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen2.5 72B Instruct","description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","context_length":32768,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Qwen","instruct_type":"chatml"},"pricing":{"prompt":"0.00000036","completion":"0.0000004"},"top_provider":{"context_length":32768,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/qwen/qwen-2.5-72b-instruct"}},{"id":"cohere/command-r-08-2024","object":"model","created":1724976000,"owned_by":"cohere","canonical_slug":"cohere/command-r-08-2024","hugging_face_id":null,"name":"Cohere: Command R (08-2024)","description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006"},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/v1/models/cohere/command-r-08-2024"}},{"id":"cohere/command-r-plus-08-2024","object":"model","created":1724976000,"owned_by":"cohere","canonical_slug":"cohere/command-r-plus-08-2024","hugging_face_id":null,"name":"Cohere: Command R+ (08-2024)","description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","context_length":128000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Cohere","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001"},"top_provider":{"context_length":128000,"max_completion_tokens":4000,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-03-31","expiration_date":null,"links":{"details":"/v1/models/cohere/command-r-plus-08-2024"}},{"id":"sao10k/l3.1-euryale-70b","object":"model","created":1724803200,"owned_by":"sao10k","canonical_slug":"sao10k/l3.1-euryale-70b","hugging_face_id":"Sao10K/L3.1-70B-Euryale-v2.2","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000085","completion":"0.00000085"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/sao10k/l3.1-euryale-70b"}},{"id":"nousresearch/hermes-3-llama-3.1-70b","object":"model","created":1723939200,"owned_by":"nousresearch","canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-70B","name":"Nous: Hermes 3 70B Instruct","description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.0000007","completion":"0.0000007"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/nousresearch/hermes-3-llama-3.1-70b"}},{"id":"nousresearch/hermes-3-llama-3.1-405b","object":"model","created":1723766400,"owned_by":"nousresearch","canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-405B","name":"Nous: Hermes 3 405B Instruct","description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"chatml"},"pricing":{"prompt":"0.000001","completion":"0.000001"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/nousresearch/hermes-3-llama-3.1-405b"}},{"id":"sao10k/l3-lunaris-8b","object":"model","created":1723507200,"owned_by":"sao10k","canonical_slug":"sao10k/l3-lunaris-8b","hugging_face_id":"Sao10K/L3-8B-Lunaris-v1","name":"Sao10K: Llama 3 8B Lunaris","description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000004","completion":"0.00000005"},"top_provider":{"context_length":8192,"max_completion_tokens":7372,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/sao10k/l3-lunaris-8b"}},{"id":"openai/gpt-4o-2024-08-06","object":"model","created":1722902400,"owned_by":"openai","canonical_slug":"openai/gpt-4o-2024-08-06","hugging_face_id":null,"name":"OpenAI: GPT-4o (2024-08-06)","description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-2024-08-06"}},{"id":"meta-llama/llama-3.1-70b-instruct","object":"model","created":1721692800,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-3.1-70b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-70B-Instruct","name":"Meta: Llama 3.1 70B Instruct","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.0000004","completion":"0.0000004"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-3.1-70b-instruct"}},{"id":"meta-llama/llama-3.1-8b-instruct","object":"model","created":1721692800,"owned_by":"meta-llama","canonical_slug":"meta-llama/llama-3.1-8b-instruct","hugging_face_id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Meta: Llama 3.1 8B Instruct","description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama3","instruct_type":"llama3"},"pricing":{"prompt":"0.00000005","completion":"0.00000008","input_cache_read":"0.000000025"},"top_provider":{"context_length":131072,"max_completion_tokens":117964,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/meta-llama/llama-3.1-8b-instruct"}},{"id":"mistralai/mistral-nemo","object":"model","created":1721347200,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-nemo","hugging_face_id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral: Mistral Nemo","description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","context_length":131072,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000000019","completion":"0.00000003"},"top_provider":{"context_length":131072,"max_completion_tokens":16384,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-nemo"}},{"id":"openai/gpt-4o-mini","object":"model","created":1721260800,"owned_by":"openai","canonical_slug":"openai/gpt-4o-mini","hugging_face_id":null,"name":"OpenAI: GPT-4o-mini","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-mini"}},{"id":"openai/gpt-4o-mini-2024-07-18","object":"model","created":1721260800,"owned_by":"openai","canonical_slug":"openai/gpt-4o-mini-2024-07-18","hugging_face_id":null,"name":"OpenAI: GPT-4o-mini (2024-07-18)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000015","completion":"0.0000006","input_cache_read":"0.000000075"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-mini-2024-07-18"}},{"id":"openai/gpt-4o-mini:batch","object":"model","created":1721260800,"owned_by":"openai","canonical_slug":"openai/gpt-4o-mini","hugging_face_id":null,"name":"OpenAI: GPT-4o-mini (batch)","description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000000075","completion":"0.0000003","web_search":"0.01","input_cache_read":"0.0000000375"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-mini:batch"}},{"id":"google/gemma-2-27b-it","object":"model","created":1720828800,"owned_by":"google","canonical_slug":"google/gemma-2-27b-it","hugging_face_id":"google/gemma-2-27b-it","name":"Google: Gemma 2 27B","description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Gemini","instruct_type":"gemma"},"pricing":{"prompt":"0.00000065","completion":"0.00000065"},"top_provider":{"context_length":8192,"max_completion_tokens":2048,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-06-30","expiration_date":null,"links":{"details":"/v1/models/google/gemma-2-27b-it"}},{"id":"openai/gpt-4o","object":"model","created":1715558400,"owned_by":"openai","canonical_slug":"openai/gpt-4o","hugging_face_id":null,"name":"OpenAI: GPT-4o","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000025","completion":"0.00001","input_cache_read":"0.00000125"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o"}},{"id":"openai/gpt-4o-2024-05-13","object":"model","created":1715558400,"owned_by":"openai","canonical_slug":"openai/gpt-4o-2024-05-13","hugging_face_id":null,"name":"OpenAI: GPT-4o (2024-05-13)","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000015"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o-2024-05-13"}},{"id":"openai/gpt-4o:batch","object":"model","created":1715558400,"owned_by":"openai","canonical_slug":"openai/gpt-4o","hugging_face_id":null,"name":"OpenAI: GPT-4o (batch)","description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","context_length":128000,"architecture":{"modality":"text+image+file->text","input_modalities":["text","image","file"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000125","completion":"0.000005","web_search":"0.01","input_cache_read":"0.000000625"},"top_provider":{"context_length":128000,"max_completion_tokens":16384,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","prediction","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-10-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4o:batch"}},{"id":"mistralai/mixtral-8x22b-instruct","object":"model","created":1713312000,"owned_by":"mistralai","canonical_slug":"mistralai/mixtral-8x22b-instruct","hugging_face_id":"mistralai/Mixtral-8x22B-Instruct-v0.1","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","context_length":65536,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"mistral"},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":65536,"max_completion_tokens":52428,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-01-31","expiration_date":null,"links":{"details":"/v1/models/mistralai/mixtral-8x22b-instruct"}},{"id":"microsoft/wizardlm-2-8x22b","object":"model","created":1713225600,"owned_by":"microsoft","canonical_slug":"microsoft/wizardlm-2-8x22b","hugging_face_id":"microsoft/WizardLM-2-8x22B","name":"WizardLM-2 8x22B","description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","context_length":65535,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":"vicuna"},"pricing":{"prompt":"0.00000062","completion":"0.00000062"},"top_provider":{"context_length":65535,"max_completion_tokens":8000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2024-04-30","expiration_date":null,"links":{"details":"/v1/models/microsoft/wizardlm-2-8x22b"}},{"id":"openai/gpt-4-turbo","object":"model","created":1712620800,"owned_by":"openai","canonical_slug":"openai/gpt-4-turbo","hugging_face_id":null,"name":"OpenAI: GPT-4 Turbo","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00001","completion":"0.00003"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4-turbo"}},{"id":"openai/gpt-4-turbo:batch","object":"model","created":1712620800,"owned_by":"openai","canonical_slug":"openai/gpt-4-turbo","hugging_face_id":null,"name":"OpenAI: GPT-4 Turbo (batch)","description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","context_length":128000,"architecture":{"modality":"text+image->text","input_modalities":["text","image"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000005","completion":"0.000015","web_search":"0.01"},"top_provider":{"context_length":128000,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-12-31","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4-turbo:batch"}},{"id":"mistralai/mistral-large","object":"model","created":1708905600,"owned_by":"mistralai","canonical_slug":"mistralai/mistral-large","hugging_face_id":null,"name":"Mistral Large","description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","context_length":128000,"architecture":{"modality":"text+file->text","input_modalities":["text","file"],"output_modalities":["text"],"tokenizer":"Mistral","instruct_type":null},"pricing":{"prompt":"0.000002","completion":"0.000006","input_cache_read":"0.0000002"},"top_provider":{"context_length":128000,"max_completion_tokens":102400,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"default_parameters":{"temperature":0.3},"supported_voices":null,"knowledge_cutoff":"2024-11-30","expiration_date":null,"links":{"details":"/v1/models/mistralai/mistral-large"}},{"id":"openai/gpt-3.5-turbo-0613","object":"model","created":1706140800,"owned_by":"openai","canonical_slug":"openai/gpt-3.5-turbo-0613","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo (older v0613)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000001","completion":"0.000002"},"top_provider":{"context_length":4095,"max_completion_tokens":3685,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-3.5-turbo-0613"}},{"id":"openai/gpt-3.5-turbo-instruct","object":"model","created":1695859200,"owned_by":"openai","canonical_slug":"openai/gpt-3.5-turbo-instruct","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo Instruct","description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","context_length":4095,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":"chatml"},"pricing":{"prompt":"0.0000015","completion":"0.000002"},"top_provider":{"context_length":4095,"max_completion_tokens":3685,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-3.5-turbo-instruct"}},{"id":"openai/gpt-3.5-turbo-16k","object":"model","created":1693180800,"owned_by":"openai","canonical_slug":"openai/gpt-3.5-turbo-16k","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo 16k","description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.000003","completion":"0.000004"},"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-3.5-turbo-16k"}},{"id":"mancer/weaver","object":"model","created":1690934400,"owned_by":"mancer","canonical_slug":"mancer/weaver","hugging_face_id":null,"name":"Mancer: Weaver (alpha)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","context_length":8000,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.0000004","completion":"0.00000075"},"top_provider":{"context_length":8000,"max_completion_tokens":6000,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/v1/models/mancer/weaver"}},{"id":"undi95/remm-slerp-l2-13b","object":"model","created":1689984000,"owned_by":"undi95","canonical_slug":"undi95/remm-slerp-l2-13b","hugging_face_id":"Undi95/ReMM-SLERP-L2-13B","name":"ReMM SLERP 13B","description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","context_length":6144,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.00000035","completion":"0.00000065"},"top_provider":{"context_length":6144,"max_completion_tokens":5529,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/v1/models/undi95/remm-slerp-l2-13b"}},{"id":"gryphe/mythomax-l2-13b","object":"model","created":1688256000,"owned_by":"gryphe","canonical_slug":"gryphe/mythomax-l2-13b","hugging_face_id":"Gryphe/MythoMax-L2-13b","name":"MythoMax 13B","description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","context_length":8192,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"Llama2","instruct_type":"alpaca"},"pricing":{"prompt":"0.00000008","completion":"0.00000011"},"top_provider":{"context_length":4096,"max_completion_tokens":3686,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2023-06-30","expiration_date":null,"links":{"details":"/v1/models/gryphe/mythomax-l2-13b"}},{"id":"openai/gpt-3.5-turbo","object":"model","created":1685232000,"owned_by":"openai","canonical_slug":"openai/gpt-3.5-turbo","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.0000005","completion":"0.0000015"},"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-3.5-turbo"}},{"id":"openai/gpt-3.5-turbo:batch","object":"model","created":1685232000,"owned_by":"openai","canonical_slug":"openai/gpt-3.5-turbo","hugging_face_id":null,"name":"OpenAI: GPT-3.5 Turbo (batch)","description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","context_length":16385,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00000025","completion":"0.00000075","web_search":"0.01"},"top_provider":{"context_length":16385,"max_completion_tokens":4096,"is_moderated":true},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-3.5-turbo:batch"}},{"id":"openai/gpt-4","object":"model","created":1685232000,"owned_by":"openai","canonical_slug":"openai/gpt-4","hugging_face_id":null,"name":"OpenAI: GPT-4","description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","context_length":8191,"architecture":{"modality":"text->text","input_modalities":["text"],"output_modalities":["text"],"tokenizer":"GPT","instruct_type":null},"pricing":{"prompt":"0.00003","completion":"0.00006"},"top_provider":{"context_length":8191,"max_completion_tokens":4096,"is_moderated":false},"per_request_limits":null,"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"default_parameters":{},"supported_voices":null,"knowledge_cutoff":"2021-09-30","expiration_date":null,"links":{"details":"/v1/models/openai/gpt-4"}}],"total_count":639,"links":{"next":null}}