{"data":[{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/cobuddy-20260430","context_length":131072,"created":1778035480,"default_parameters":{},"description":"CoBuddy is a code generation model from Baidu, optimized for coding tasks and AI Agent workflows. It features high inference throughput and low end-to-end latency, with native support for tool...","expiration_date":null,"hugging_face_id":null,"id":"baidu/cobuddy:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/baidu/cobuddy-20260430/endpoints"},"name":"Baidu Qianfan: CoBuddy (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","tools"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-chat-latest-20260505","context_length":400000,"created":1778000212,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-chat-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-chat-latest-20260505/endpoints"},"name":"OpenAI: GPT Chat Latest","per_request_limits":null,"pricing":{"completion":"0.00003","input_cache_read":"0.0000005","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","tool_choice","tools","top_logprobs"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-4.3-20260430","context_length":1000000,"created":1777591821,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Grok 4.3 is a reasoning model from xAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...","expiration_date":null,"hugging_face_id":null,"id":"x-ai/grok-4.3","knowledge_cutoff":null,"links":{"details":"/api/v1/models/x-ai/grok-4.3-20260430/endpoints"},"name":"xAI: Grok 4.3","per_request_limits":null,"pricing":{"completion":"0.0000025","input_cache_read":"0.0000002","prompt":"0.00000125","web_search":"0.005"},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"ibm-granite/granite-4.1-8b-20260429","context_length":131072,"created":1777577071,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...","expiration_date":null,"hugging_face_id":"ibm-granite/granite-4.1-8b","id":"ibm-granite/granite-4.1-8b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/ibm-granite/granite-4.1-8b-20260429/endpoints"},"name":"IBM: Granite 4.1 8B","per_request_limits":null,"pricing":{"completion":"0.0000001","input_cache_read":"0.00000005","prompt":"0.00000005"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-medium-3.5-20260430","context_length":262144,"created":1777570439,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...","expiration_date":null,"hugging_face_id":null,"id":"mistralai/mistral-medium-3-5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/mistral-medium-3.5-20260430/endpoints"},"name":"Mistral: Mistral Medium 3.5","per_request_limits":null,"pricing":{"completion":"0.0000075","prompt":"0.0000015"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"openrouter/owl-alpha","context_length":1048756,"created":1777398589,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Owl Alpha is a high-performance foundation model designed for agentic workloads. Natively supports tool use, and long-context tasks, with strong performance in code generation, automated workflows, and complex instruction execution....","expiration_date":null,"hugging_face_id":null,"id":"openrouter/owl-alpha","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openrouter/owl-alpha/endpoints"},"name":"Owl Alpha","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1048756,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text","audio","image","video"],"instruct_type":null,"modality":"text+image+audio+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428","context_length":256000,"created":1777393095,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":null,"top_p":0.95},"description":"NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...","expiration_date":null,"hugging_face_id":null,"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning-20260428/endpoints"},"name":"NVIDIA: Nemotron 3 Nano Omni (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"poolside/laguna-xs.2-20260421","context_length":131072,"created":1777389604,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.7,"top_k":null,"top_p":0.9},"description":"Laguna XS.2 is the second-generation model in the XS size class from [Poolside](https://poolside.ai), their efficient coding agent series. It combines tool calling and reasoning capabilities with a compact footprint, offering...","expiration_date":null,"hugging_face_id":"poolside/Laguna-XS.2","id":"poolside/laguna-xs.2:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/poolside/laguna-xs.2-20260421/endpoints"},"name":"Poolside: Laguna XS.2 (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"poolside/laguna-m.1-20260312","context_length":131072,"created":1777388504,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Laguna M.1 is the flagship coding agent model from [Poolside](https://poolside.ai), optimized for complex software engineering tasks. Designed for agentic coding workflows, it supports tool calling and reasoning, with a 128K...","expiration_date":null,"hugging_face_id":null,"id":"poolside/laguna-m.1:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/poolside/laguna-m.1-20260312/endpoints"},"name":"Poolside: Laguna M.1 (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~anthropic/claude-haiku-latest","context_length":200000,"created":1777318492,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the Anthropic Claude Haiku family.","expiration_date":null,"hugging_face_id":null,"id":"~anthropic/claude-haiku-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/~anthropic/claude-haiku-latest/endpoints"},"name":"Anthropic Claude Haiku Latest","per_request_limits":null,"pricing":{"completion":"0.000005","input_cache_read":"0.0000001","input_cache_write":"0.00000125","prompt":"0.000001","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~openai/gpt-mini-latest","context_length":400000,"created":1777318471,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the OpenAI GPT Mini family.","expiration_date":null,"hugging_face_id":null,"id":"~openai/gpt-mini-latest","knowledge_cutoff":"2025-08-31","links":{"details":"/api/v1/models/~openai/gpt-mini-latest/endpoints"},"name":"OpenAI GPT Mini Latest","per_request_limits":null,"pricing":{"completion":"0.0000045","input_cache_read":"0.000000075","prompt":"0.00000075","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["audio","file","image","text","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~google/gemini-pro-latest","context_length":1048576,"created":1777318451,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the Google Gemini Pro family.","expiration_date":null,"hugging_face_id":null,"id":"~google/gemini-pro-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/~google/gemini-pro-latest/endpoints"},"name":"Google Gemini Pro Latest","per_request_limits":null,"pricing":{"audio":"0.000002","completion":"0.000012","image":"0.000002","input_cache_read":"0.0000002","input_cache_write":"0.000000375","internal_reasoning":"0.000012","prompt":"0.000002","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~moonshotai/kimi-latest","context_length":262144,"created":1777318428,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the MoonshotAI Kimi family.","expiration_date":null,"hugging_face_id":null,"id":"~moonshotai/kimi-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/~moonshotai/kimi-latest/endpoints"},"name":"MoonshotAI Kimi Latest","per_request_limits":null,"pricing":{"completion":"0.0000035","input_cache_read":"0.00000015","prompt":"0.00000075"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~google/gemini-flash-latest","context_length":1048576,"created":1777318398,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the Google Gemini Flash family.","expiration_date":null,"hugging_face_id":null,"id":"~google/gemini-flash-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/~google/gemini-flash-latest/endpoints"},"name":"Google Gemini Flash Latest","per_request_limits":null,"pricing":{"audio":"0.000001","completion":"0.000003","image":"0.0000005","input_cache_read":"0.00000005","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.000003","prompt":"0.0000005","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~anthropic/claude-sonnet-latest","context_length":1000000,"created":1777318368,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the Anthropic Claude Sonnet family.","expiration_date":null,"hugging_face_id":null,"id":"~anthropic/claude-sonnet-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/~anthropic/claude-sonnet-latest/endpoints"},"name":"Anthropic Claude Sonnet Latest","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.0000003","input_cache_write":"0.00000375","prompt":"0.000003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~openai/gpt-latest","context_length":1050000,"created":1777318334,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the OpenAI GPT family.","expiration_date":null,"hugging_face_id":null,"id":"~openai/gpt-latest","knowledge_cutoff":"2025-12-01","links":{"details":"/api/v1/models/~openai/gpt-latest/endpoints"},"name":"OpenAI GPT Latest","per_request_limits":null,"pricing":{"completion":"0.00003","input_cache_read":"0.0000005","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":1050000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-plus-20260420","context_length":1000000,"created":1777261368,"default_parameters":{},"description":"Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...","expiration_date":null,"hugging_face_id":null,"id":"qwen/qwen3.5-plus-20260420","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-plus-20260420/endpoints"},"name":"Qwen: Qwen3.5 Plus 2026-04-20","per_request_limits":null,"pricing":{"completion":"0.0000024","prompt":"0.0000004"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.6-flash","context_length":1000000,"created":1777261362,"default_parameters":{},"description":"Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...","expiration_date":null,"hugging_face_id":null,"id":"qwen/qwen3.6-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-flash/endpoints"},"name":"Qwen: Qwen3.6 Flash","per_request_limits":null,"pricing":{"completion":"0.0000015","input_cache_write":"0.0000003125","prompt":"0.00000025"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen3.6-35b-a3b-20260415","context_length":262144,"created":1777260255,"default_parameters":{"temperature":1,"top_k":20,"top_p":0.95},"description":"Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.6-35B-A3B","id":"qwen/qwen3.6-35b-a3b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-35b-a3b-20260415/endpoints"},"name":"Qwen: Qwen3.6 35B A3B","per_request_limits":null,"pricing":{"completion":"0.000001","input_cache_read":"0.00000005","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen3.6-max-preview-20260420","context_length":262144,"created":1777260242,"default_parameters":{},"description":"Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...","expiration_date":null,"hugging_face_id":null,"id":"qwen/qwen3.6-max-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-max-preview-20260420/endpoints"},"name":"Qwen: Qwen3.6 Max Preview","per_request_limits":null,"pricing":{"completion":"0.00000624","input_cache_write":"0.0000013","prompt":"0.00000104"},"supported_parameters":["include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.6-27b-20260422","context_length":262144,"created":1777255064,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.6-27B","id":"qwen/qwen3.6-27b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-27b-20260422/endpoints"},"name":"Qwen: Qwen3.6 27B","per_request_limits":null,"pricing":{"completion":"0.0000032","prompt":"0.00000032"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":81920}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.5-pro-20260423","context_length":1050000,"created":1777051896,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.5-pro","knowledge_cutoff":"2025-12-01","links":{"details":"/api/v1/models/openai/gpt-5.5-pro-20260423/endpoints"},"name":"OpenAI: GPT-5.5 Pro","per_request_limits":null,"pricing":{"completion":"0.00018","prompt":"0.00003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":1050000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.5-20260423","context_length":1050000,"created":1777051893,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.5","knowledge_cutoff":"2025-12-01","links":{"details":"/api/v1/models/openai/gpt-5.5-20260423/endpoints"},"name":"OpenAI: GPT-5.5","per_request_limits":null,"pricing":{"completion":"0.00003","input_cache_read":"0.0000005","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":1050000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-v4-pro-20260423","context_length":1048576,"created":1777000679,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":1},"description":"DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V4-Pro","id":"deepseek/deepseek-v4-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-pro-20260423/endpoints"},"name":"DeepSeek: DeepSeek V4 Pro","per_request_limits":null,"pricing":{"completion":"0.00000087","input_cache_read":"0.000000003625","prompt":"0.000000435"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":384000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-v4-flash-20260423","context_length":1048576,"created":1777000666,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V4-Flash","id":"deepseek/deepseek-v4-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v4-flash-20260423/endpoints"},"name":"DeepSeek: DeepSeek V4 Flash","per_request_limits":null,"pricing":{"completion":"0.00000028","input_cache_read":"0.0000000028","prompt":"0.00000014"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":384000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"inclusionai/ling-2.6-1t-20260423","context_length":262144,"created":1776948238,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...","expiration_date":"2026-05-07","hugging_face_id":null,"id":"inclusionai/ling-2.6-1t:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/inclusionai/ling-2.6-1t-20260423/endpoints"},"name":"inclusionAI: Ling-2.6-1T (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"tencent/hy3-preview-20260421","context_length":262144,"created":1776878150,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.9,"top_k":null,"top_p":1},"description":"Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...","expiration_date":null,"hugging_face_id":"tencent/Hy3-preview","id":"tencent/hy3-preview:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/tencent/hy3-preview-20260421/endpoints"},"name":"Tencent: Hy3 preview (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"xiaomi/mimo-v2.5-pro-20260422","context_length":1048576,"created":1776874273,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....","expiration_date":null,"hugging_face_id":"XiaomiMiMo/MiMo-V2.5-Pro","id":"xiaomi/mimo-v2.5-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2.5-pro-20260422/endpoints"},"name":"Xiaomi: MiMo-V2.5-Pro","per_request_limits":null,"pricing":{"completion":"0.000003","input_cache_read":"0.0000002","prompt":"0.000001"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text","audio","image","video"],"instruct_type":null,"modality":"text+image+audio+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"xiaomi/mimo-v2.5-20260422","context_length":1048576,"created":1776874269,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...","expiration_date":null,"hugging_face_id":"XiaomiMiMo/MiMo-V2.5","id":"xiaomi/mimo-v2.5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2.5-20260422/endpoints"},"name":"Xiaomi: MiMo-V2.5","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000008","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text+image","output_modalities":["image","text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.4-image-2-20260421","context_length":272000,"created":1776797528,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.4-image-2","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-image-2-20260421/endpoints"},"name":"OpenAI: GPT-5.4 Image 2","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.000002","prompt":"0.000008","web_search":"0.01"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","top_logprobs"],"supported_voices":null,"top_provider":{"context_length":272000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"inclusionai/ling-2.6-flash-20260421","context_length":262144,"created":1776795886,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....","expiration_date":null,"hugging_face_id":"","id":"inclusionai/ling-2.6-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/inclusionai/ling-2.6-flash-20260421/endpoints"},"name":"inclusionAI: Ling-2.6-flash","per_request_limits":null,"pricing":{"completion":"0.00000024","input_cache_read":"0.000000016","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"~anthropic/claude-opus-latest","context_length":1000000,"created":1776795361,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"This model always redirects to the latest model in the Claude Opus family.","expiration_date":null,"hugging_face_id":"","id":"~anthropic/claude-opus-latest","knowledge_cutoff":null,"links":{"details":"/api/v1/models/~anthropic/claude-opus-latest/endpoints"},"name":"Anthropic: Claude Opus Latest","per_request_limits":null,"pricing":{"completion":"0.000025","input_cache_read":"0.0000005","input_cache_write":"0.00000625","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"openrouter/pareto-code","context_length":200000,"created":1776747900,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"The Pareto Router is a way to have OpenRouter always pick a strong coding model for your needs without committing to a specific one. You express a single `min_coding_score` preference...","expiration_date":null,"hugging_face_id":"","id":"openrouter/pareto-code","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openrouter/pareto-code/endpoints"},"name":"Pareto Code Router","per_request_limits":null,"pricing":{"completion":"-1","prompt":"-1"},"supported_parameters":[],"supported_voices":null,"top_provider":{"context_length":null,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/qianfan-ocr-fast-20260420","context_length":65536,"created":1776707472,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Qianfan-OCR-Fast is a domain-specific multimodal large model purpose-built for OCR. By leveraging specialized OCR training data while preserving versatile multimodal intelligence, it provides a powerful performance upgrade over Qianfan-OCR.","expiration_date":null,"hugging_face_id":"","id":"baidu/qianfan-ocr-fast:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/baidu/qianfan-ocr-fast-20260420/endpoints"},"name":"Baidu: Qianfan-OCR-Fast (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":28672}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"moonshotai/kimi-k2.6-20260420","context_length":262144,"created":1776699402,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...","expiration_date":null,"hugging_face_id":"moonshotai/Kimi-K2.6","id":"moonshotai/kimi-k2.6","knowledge_cutoff":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2.6-20260420/endpoints"},"name":"MoonshotAI: Kimi K2.6","per_request_limits":null,"pricing":{"completion":"0.0000035","input_cache_read":"0.00000015","prompt":"0.00000075"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.7-opus-20260416","context_length":1000000,"created":1776351100,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...","expiration_date":null,"hugging_face_id":null,"id":"anthropic/claude-opus-4.7","knowledge_cutoff":null,"links":{"details":"/api/v1/models/anthropic/claude-4.7-opus-20260416/endpoints"},"name":"Anthropic: Claude Opus 4.7","per_request_limits":null,"pricing":{"completion":"0.000025","input_cache_read":"0.0000005","input_cache_write":"0.00000625","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","tool_choice","tools","verbosity"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.6-opus-fast-20260407","context_length":1000000,"created":1775592472,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Fast-mode variant of [Opus 4.6](/anthropic/claude-opus-4.6) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode","expiration_date":null,"hugging_face_id":null,"id":"anthropic/claude-opus-4.6-fast","knowledge_cutoff":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-opus-fast-20260407/endpoints"},"name":"Anthropic: Claude Opus 4.6 (Fast)","per_request_limits":null,"pricing":{"completion":"0.00015","input_cache_read":"0.000003","input_cache_write":"0.0000375","prompt":"0.00003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p","verbosity"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-5.1-20260406","context_length":202752,"created":1775578025,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...","expiration_date":null,"hugging_face_id":"zai-org/GLM-5.1","id":"z-ai/glm-5.1","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-5.1-20260406/endpoints"},"name":"Z.ai: GLM 5.1","per_request_limits":null,"pricing":{"completion":"0.0000035","input_cache_read":"0.000000525","prompt":"0.00000105"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":202752,"is_moderated":false,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Gemma"},"canonical_slug":"google/gemma-4-26b-a4b-it-20260403","context_length":262144,"created":1775227989,"default_parameters":{"temperature":1,"top_k":64,"top_p":0.95},"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","expiration_date":null,"hugging_face_id":"google/gemma-4-26B-A4B-it","id":"google/gemma-4-26b-a4b-it:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints"},"name":"Google: Gemma 4 26B A4B  (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Gemma"},"canonical_slug":"google/gemma-4-26b-a4b-it-20260403","context_length":262144,"created":1775227989,"default_parameters":{"temperature":1,"top_k":64,"top_p":0.95},"description":"Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...","expiration_date":null,"hugging_face_id":"google/gemma-4-26B-A4B-it","id":"google/gemma-4-26b-a4b-it","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemma-4-26b-a4b-it-20260403/endpoints"},"name":"Google: Gemma 4 26B A4B ","per_request_limits":null,"pricing":{"completion":"0.00000033","prompt":"0.00000006"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Gemma"},"canonical_slug":"google/gemma-4-31b-it-20260402","context_length":262144,"created":1775148486,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":64,"top_p":0.95},"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","expiration_date":null,"hugging_face_id":"google/gemma-4-31B-it","id":"google/gemma-4-31b-it:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemma-4-31b-it-20260402/endpoints"},"name":"Google: Gemma 4 31B (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Gemma"},"canonical_slug":"google/gemma-4-31b-it-20260402","context_length":262144,"created":1775148486,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":64,"top_p":0.95},"description":"Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...","expiration_date":null,"hugging_face_id":"google/gemma-4-31B-it","id":"google/gemma-4-31b-it","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemma-4-31b-it-20260402/endpoints"},"name":"Google: Gemma 4 31B","per_request_limits":null,"pricing":{"completion":"0.00000038","prompt":"0.00000013"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.6-plus-04-02","context_length":1000000,"created":1775133557,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen3.6-plus","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.6-plus-04-02/endpoints"},"name":"Qwen: Qwen3.6 Plus","per_request_limits":null,"pricing":{"completion":"0.00000195","input_cache_write":"0.00000040625","prompt":"0.000000325"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-5v-turbo-20260401","context_length":202752,"created":1775061458,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...","expiration_date":null,"hugging_face_id":"","id":"z-ai/glm-5v-turbo","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-5v-turbo-20260401/endpoints"},"name":"Z.ai: GLM 5V Turbo","per_request_limits":null,"pricing":{"completion":"0.000004","input_cache_read":"0.00000024","prompt":"0.0000012"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":202752,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/trinity-large-thinking","context_length":262144,"created":1775058318,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.3,"top_k":null,"top_p":0.8},"description":"Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7","expiration_date":null,"hugging_face_id":"arcee-ai/Trinity-Large-Thinking","id":"arcee-ai/trinity-large-thinking","knowledge_cutoff":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-large-thinking/endpoints"},"name":"Arcee AI: Trinity Large Thinking","per_request_limits":null,"pricing":{"completion":"0.00000085","input_cache_read":"0.00000006","prompt":"0.00000022"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-4.20-multi-agent-20260309","context_length":2000000,"created":1774979158,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Grok 4.20 Multi-Agent is a variant of xAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...","expiration_date":null,"hugging_face_id":"","id":"x-ai/grok-4.20-multi-agent","knowledge_cutoff":"2025-09-01","links":{"details":"/api/v1/models/x-ai/grok-4.20-multi-agent-20260309/endpoints"},"name":"xAI: Grok 4.20 Multi-Agent","per_request_limits":null,"pricing":{"completion":"0.000006","input_cache_read":"0.0000002","prompt":"0.000002","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":2000000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-4.20-20260309","context_length":2000000,"created":1774979019,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Grok 4.20 is xAI's newest flagship model with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering consistently...","expiration_date":null,"hugging_face_id":"","id":"x-ai/grok-4.20","knowledge_cutoff":"2025-09-01","links":{"details":"/api/v1/models/x-ai/grok-4.20-20260309/endpoints"},"name":"xAI: Grok 4.20","per_request_limits":null,"pricing":{"completion":"0.0000025","input_cache_read":"0.0000002","prompt":"0.00000125","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":2000000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text+audio","output_modalities":["text","audio"],"tokenizer":"Other"},"canonical_slug":"google/lyria-3-pro-preview-20260330","context_length":1048576,"created":1774907286,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...","expiration_date":null,"hugging_face_id":null,"id":"google/lyria-3-pro-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/lyria-3-pro-preview-20260330/endpoints"},"name":"Google: Lyria 3 Pro Preview","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text+audio","output_modalities":["text","audio"],"tokenizer":"Other"},"canonical_slug":"google/lyria-3-clip-preview-20260330","context_length":1048576,"created":1774907255,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...","expiration_date":null,"hugging_face_id":null,"id":"google/lyria-3-clip-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/lyria-3-clip-preview-20260330/endpoints"},"name":"Google: Lyria 3 Clip Preview","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["max_tokens","response_format","seed","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"kwaipilot/kat-coder-pro-v2-20260327","context_length":256000,"created":1774649310,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...","expiration_date":null,"hugging_face_id":"","id":"kwaipilot/kat-coder-pro-v2","knowledge_cutoff":null,"links":{"details":"/api/v1/models/kwaipilot/kat-coder-pro-v2-20260327/endpoints"},"name":"Kwaipilot: KAT-Coder-Pro V2","per_request_limits":null,"pricing":{"completion":"0.0000012","input_cache_read":"0.00000006","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":80000}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"rekaai/reka-edge-2603","context_length":16384,"created":1774026965,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...","expiration_date":null,"hugging_face_id":"RekaAI/reka-edge-2603","id":"rekaai/reka-edge","knowledge_cutoff":null,"links":{"details":"/api/v1/models/rekaai/reka-edge-2603/endpoints"},"name":"Reka Edge","per_request_limits":null,"pricing":{"completion":"0.0000001","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":16384,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","audio","image","video"],"instruct_type":null,"modality":"text+image+audio+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"xiaomi/mimo-v2-omni-20260318","context_length":262144,"created":1773863703,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiMo-V2-Omni is a frontier omni-modal model that natively processes image, video, and audio inputs within a unified architecture. It combines strong multimodal perception with agentic capability - visual grounding, multi-step...","expiration_date":null,"hugging_face_id":"","id":"xiaomi/mimo-v2-omni","knowledge_cutoff":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2-omni-20260318/endpoints"},"name":"Xiaomi: MiMo-V2-Omni","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000008","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"xiaomi/mimo-v2-pro-20260318","context_length":1048576,"created":1773863643,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiMo-V2-Pro is Xiaomi's flagship foundation model, featuring over 1T total parameters and a 1M context length, deeply optimized for agentic scenarios. It is highly adaptable to general agent frameworks like...","expiration_date":null,"hugging_face_id":"","id":"xiaomi/mimo-v2-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2-pro-20260318/endpoints"},"name":"Xiaomi: MiMo-V2-Pro","per_request_limits":null,"pricing":{"completion":"0.000003","input_cache_read":"0.0000002","prompt":"0.000001"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m2.7-20260318","context_length":196608,"created":1773836697,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...","expiration_date":null,"hugging_face_id":"MiniMaxAI/MiniMax-M2.7","id":"minimax/minimax-m2.7","knowledge_cutoff":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.7-20260318/endpoints"},"name":"MiniMax: MiniMax M2.7","per_request_limits":null,"pricing":{"completion":"0.0000012","input_cache_read":"0.000000059","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":196608,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.4-nano-20260317","context_length":400000,"created":1773748187,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.4-nano","knowledge_cutoff":"2025-08-31","links":{"details":"/api/v1/models/openai/gpt-5.4-nano-20260317/endpoints"},"name":"OpenAI: GPT-5.4 Nano","per_request_limits":null,"pricing":{"completion":"0.00000125","input_cache_read":"0.00000002","prompt":"0.0000002","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.4-mini-20260317","context_length":400000,"created":1773748178,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.4-mini","knowledge_cutoff":"2025-08-31","links":{"details":"/api/v1/models/openai/gpt-5.4-mini-20260317/endpoints"},"name":"OpenAI: GPT-5.4 Mini","per_request_limits":null,"pricing":{"completion":"0.0000045","input_cache_read":"0.000000075","prompt":"0.00000075","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-small-2603","context_length":262144,"created":1773695685,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...","expiration_date":null,"hugging_face_id":"mistralai/Mistral-Small-4-119B-2603","id":"mistralai/mistral-small-2603","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/mistral-small-2603/endpoints"},"name":"Mistral: Mistral Small 4","per_request_limits":null,"pricing":{"completion":"0.0000006","input_cache_read":"0.000000015","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-5-turbo-20260315","context_length":202752,"created":1773583573,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...","expiration_date":null,"hugging_face_id":"","id":"z-ai/glm-5-turbo","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-5-turbo-20260315/endpoints"},"name":"Z.ai: GLM 5 Turbo","per_request_limits":null,"pricing":{"completion":"0.000004","input_cache_read":"0.00000024","prompt":"0.0000012"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":202752,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","context_length":262144,"created":1773245239,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","id":"nvidia/nemotron-3-super-120b-a12b:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints"},"name":"NVIDIA: Nemotron 3 Super (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-3-super-120b-a12b-20230311","context_length":262144,"created":1773245239,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","id":"nvidia/nemotron-3-super-120b-a12b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-super-120b-a12b-20230311/endpoints"},"name":"NVIDIA: Nemotron 3 Super","per_request_limits":null,"pricing":{"completion":"0.00000045","prompt":"0.00000009"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"bytedance-seed/seed-2.0-lite-20260309","context_length":262144,"created":1773157231,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...","expiration_date":null,"hugging_face_id":null,"id":"bytedance-seed/seed-2.0-lite","knowledge_cutoff":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-2.0-lite-20260309/endpoints"},"name":"ByteDance Seed: Seed-2.0-Lite","per_request_limits":null,"pricing":{"completion":"0.000002","prompt":"0.00000025"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-9b-20260310","context_length":262144,"created":1773152396,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.5-9B","id":"qwen/qwen3.5-9b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-9b-20260310/endpoints"},"name":"Qwen: Qwen3.5-9B","per_request_limits":null,"pricing":{"completion":"0.00000015","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.4-pro-20260305","context_length":1050000,"created":1772734366,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.4-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-pro-20260305/endpoints"},"name":"OpenAI: GPT-5.4 Pro","per_request_limits":null,"pricing":{"completion":"0.00018","prompt":"0.00003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":1050000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.4-20260305","context_length":1050000,"created":1772734352,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.4","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.4-20260305/endpoints"},"name":"OpenAI: GPT-5.4","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.00000025","prompt":"0.0000025","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":1050000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"inception/mercury-2-20260304","context_length":128000,"created":1772636275,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.75,"top_k":null,"top_p":null},"description":"Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...","expiration_date":null,"hugging_face_id":null,"id":"inception/mercury-2","knowledge_cutoff":null,"links":{"details":"/api/v1/models/inception/mercury-2-20260304/endpoints"},"name":"Inception: Mercury 2","per_request_limits":null,"pricing":{"completion":"0.00000075","input_cache_read":"0.000000025","prompt":"0.00000025"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":50000}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.3-chat-20260303","context_length":128000,"created":1772564061,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.3-chat","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.3-chat-20260303/endpoints"},"name":"OpenAI: GPT-5.3 Chat","per_request_limits":null,"pricing":{"completion":"0.000014","input_cache_read":"0.000000175","prompt":"0.00000175","web_search":"0.01"},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","video","file","audio"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-3.1-flash-lite-preview-20260303","context_length":1048576,"created":1772512673,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-3.1-flash-lite-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemini-3.1-flash-lite-preview-20260303/endpoints"},"name":"Google: Gemini 3.1 Flash Lite Preview","per_request_limits":null,"pricing":{"audio":"0.0000005","completion":"0.0000015","image":"0.00000025","input_cache_read":"0.000000025","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.0000015","prompt":"0.00000025","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"bytedance-seed/seed-2.0-mini-20260224","context_length":262144,"created":1772131107,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...","expiration_date":null,"hugging_face_id":"","id":"bytedance-seed/seed-2.0-mini","knowledge_cutoff":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-2.0-mini-20260224/endpoints"},"name":"ByteDance Seed: Seed-2.0-Mini","per_request_limits":null,"pricing":{"completion":"0.0000004","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text+image","output_modalities":["image","text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-3.1-flash-image-preview-20260226","context_length":65536,"created":1772119558,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-3.1-flash-image-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemini-3.1-flash-image-preview-20260226/endpoints"},"name":"Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)","per_request_limits":null,"pricing":{"completion":"0.000003","prompt":"0.0000005","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-35b-a3b-20260224","context_length":262144,"created":1772053822,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":20,"top_p":0.95},"description":"The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.5-35B-A3B","id":"qwen/qwen3.5-35b-a3b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-35b-a3b-20260224/endpoints"},"name":"Qwen: Qwen3.5-35B-A3B","per_request_limits":null,"pricing":{"completion":"0.000001","input_cache_read":"0.00000005","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-27b-20260224","context_length":262144,"created":1772053810,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":20,"top_p":0.95},"description":"The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.5-27B","id":"qwen/qwen3.5-27b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-27b-20260224/endpoints"},"name":"Qwen: Qwen3.5-27B","per_request_limits":null,"pricing":{"completion":"0.00000156","prompt":"0.000000195"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-122b-a10b-20260224","context_length":262144,"created":1772053789,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":20,"top_p":0.95},"description":"The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.5-122B-A10B","id":"qwen/qwen3.5-122b-a10b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-122b-a10b-20260224/endpoints"},"name":"Qwen: Qwen3.5-122B-A10B","per_request_limits":null,"pricing":{"completion":"0.00000208","prompt":"0.00000026"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-flash-20260224","context_length":1000000,"created":1772053776,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...","expiration_date":null,"hugging_face_id":null,"id":"qwen/qwen3.5-flash-02-23","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-flash-20260224/endpoints"},"name":"Qwen: Qwen3.5-Flash","per_request_limits":null,"pricing":{"completion":"0.00000026","input_cache_write":"0.00000008125","prompt":"0.000000065"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"liquid/lfm-2-24b-a2b-20260224","context_length":32768,"created":1772048711,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1.05,"temperature":0.1,"top_k":50,"top_p":null},"description":"LFM2-24B-A2B is the largest model in the LFM2 family of hybrid architectures designed for efficient on-device deployment. Built as a 24B parameter Mixture-of-Experts model with only 2B active parameters per...","expiration_date":null,"hugging_face_id":"LiquidAI/LFM2-24B-A2B","id":"liquid/lfm-2-24b-a2b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/liquid/lfm-2-24b-a2b-20260224/endpoints"},"name":"LiquidAI: LFM2-24B-A2B","per_request_limits":null,"pricing":{"completion":"0.00000012","prompt":"0.00000003"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","audio","image","video","file"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-3.1-pro-preview-customtools-20260219","context_length":1048576,"created":1772045923,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...","expiration_date":null,"hugging_face_id":null,"id":"google/gemini-3.1-pro-preview-customtools","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemini-3.1-pro-preview-customtools-20260219/endpoints"},"name":"Google: Gemini 3.1 Pro Preview Custom Tools","per_request_limits":null,"pricing":{"audio":"0.000002","completion":"0.000012","image":"0.000002","input_cache_read":"0.0000002","input_cache_write":"0.000000375","internal_reasoning":"0.000012","prompt":"0.000002","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.3-codex-20260224","context_length":400000,"created":1771959164,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.3-codex","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.3-codex-20260224/endpoints"},"name":"OpenAI: GPT-5.3-Codex","per_request_limits":null,"pricing":{"completion":"0.000014","input_cache_read":"0.000000175","prompt":"0.00000175","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"aion-labs/aion-2.0-20260223","context_length":131072,"created":1771881306,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....","expiration_date":null,"hugging_face_id":null,"id":"aion-labs/aion-2.0","knowledge_cutoff":null,"links":{"details":"/api/v1/models/aion-labs/aion-2.0-20260223/endpoints"},"name":"AionLabs: Aion-2.0","per_request_limits":null,"pricing":{"completion":"0.0000016","input_cache_read":"0.0000002","prompt":"0.0000008"},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["audio","file","image","text","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-3.1-pro-preview-20260219","context_length":1048576,"created":1771509627,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-3.1-pro-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemini-3.1-pro-preview-20260219/endpoints"},"name":"Google: Gemini 3.1 Pro Preview","per_request_limits":null,"pricing":{"audio":"0.000002","completion":"0.000012","image":"0.000002","input_cache_read":"0.0000002","input_cache_write":"0.000000375","internal_reasoning":"0.000012","prompt":"0.000002","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.6-sonnet-20260217","context_length":1000000,"created":1771342990,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-sonnet-4.6","knowledge_cutoff":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-sonnet-20260217/endpoints"},"name":"Anthropic: Claude Sonnet 4.6","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.0000003","input_cache_write":"0.00000375","prompt":"0.000003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-plus-20260216","context_length":1000000,"created":1771229416,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen3.5-plus-02-15","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-plus-20260216/endpoints"},"name":"Qwen: Qwen3.5 Plus 2026-02-15","per_request_limits":null,"pricing":{"completion":"0.00000156","input_cache_write":"0.000000325","prompt":"0.00000026"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3.5-397b-a17b-20260216","context_length":262144,"created":1771223018,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":20,"top_p":0.95},"description":"The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3.5-397B-A17B","id":"qwen/qwen3.5-397b-a17b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3.5-397b-a17b-20260216/endpoints"},"name":"Qwen: Qwen3.5 397B A17B","per_request_limits":null,"pricing":{"completion":"0.00000234","input_cache_read":"0.000000195","prompt":"0.00000039"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m2.5-20260211","context_length":196608,"created":1770908502,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","expiration_date":null,"hugging_face_id":"MiniMaxAI/MiniMax-M2.5","id":"minimax/minimax-m2.5:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.5-20260211/endpoints"},"name":"MiniMax: MiniMax M2.5 (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","temperature","tools"],"supported_voices":null,"top_provider":{"context_length":196608,"is_moderated":true,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m2.5-20260211","context_length":196608,"created":1770908502,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...","expiration_date":null,"hugging_face_id":"MiniMaxAI/MiniMax-M2.5","id":"minimax/minimax-m2.5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.5-20260211/endpoints"},"name":"MiniMax: MiniMax M2.5","per_request_limits":null,"pricing":{"completion":"0.00000115","input_cache_read":"0.00000003","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","reasoning_effort","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":196608,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-5-20260211","context_length":202752,"created":1770829182,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...","expiration_date":null,"hugging_face_id":"zai-org/GLM-5","id":"z-ai/glm-5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-5-20260211/endpoints"},"name":"Z.ai: GLM 5","per_request_limits":null,"pricing":{"completion":"0.00000192","input_cache_read":"0.00000012","prompt":"0.0000006"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":202752,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen3-max-thinking-20260123","context_length":262144,"created":1770671901,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...","expiration_date":null,"hugging_face_id":null,"id":"qwen/qwen3-max-thinking","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3-max-thinking-20260123/endpoints"},"name":"Qwen: Qwen3 Max Thinking","per_request_limits":null,"pricing":{"completion":"0.0000039","prompt":"0.00000078"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.6-opus-20260205","context_length":1000000,"created":1770219050,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-opus-4.6","knowledge_cutoff":null,"links":{"details":"/api/v1/models/anthropic/claude-4.6-opus-20260205/endpoints"},"name":"Anthropic: Claude Opus 4.6","per_request_limits":null,"pricing":{"completion":"0.000025","input_cache_read":"0.0000005","input_cache_write":"0.00000625","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p","verbosity"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen3-coder-next-2025-02-03","context_length":262144,"created":1770164101,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Coder-Next","id":"qwen/qwen3-coder-next","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3-coder-next-2025-02-03/endpoints"},"name":"Qwen: Qwen3 Coder Next","per_request_limits":null,"pricing":{"completion":"0.0000008","input_cache_read":"0.00000007","prompt":"0.00000011"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"openrouter/free","context_length":200000,"created":1769917427,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...","expiration_date":null,"hugging_face_id":"","id":"openrouter/free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openrouter/free/endpoints"},"name":"Free Models Router","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":null,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"stepfun/step-3.5-flash","context_length":262144,"created":1769728337,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....","expiration_date":null,"hugging_face_id":"stepfun-ai/Step-3.5-Flash","id":"stepfun/step-3.5-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/stepfun/step-3.5-flash/endpoints"},"name":"StepFun: Step 3.5 Flash","per_request_limits":null,"pricing":{"completion":"0.0000003","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/trinity-large-preview","context_length":131000,"created":1769552670,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.8,"top_k":null,"top_p":0.8},"description":"Trinity-Large-Preview is a frontier-scale open-weight language model from Arcee, built as a 400B-parameter sparse Mixture-of-Experts with 13B active parameters per token using 4-of-256 expert routing. It excels in creative writing,...","expiration_date":null,"hugging_face_id":"arcee-ai/Trinity-Large-Preview","id":"arcee-ai/trinity-large-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-large-preview/endpoints"},"name":"Arcee AI: Trinity Large Preview","per_request_limits":null,"pricing":{"completion":"0.00000045","prompt":"0.00000015"},"supported_parameters":["max_tokens","response_format","structured_outputs","temperature","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"moonshotai/kimi-k2.5-0127","context_length":262144,"created":1769487076,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...","expiration_date":null,"hugging_face_id":"moonshotai/Kimi-K2.5","id":"moonshotai/kimi-k2.5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2.5-0127/endpoints"},"name":"MoonshotAI: Kimi K2.5","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000022","prompt":"0.00000044"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","parallel_tool_calls","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"upstage/solar-pro-3","context_length":128000,"created":1769481200,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...","expiration_date":null,"hugging_face_id":"","id":"upstage/solar-pro-3","knowledge_cutoff":null,"links":{"details":"/api/v1/models/upstage/solar-pro-3/endpoints"},"name":"Upstage: Solar Pro 3","per_request_limits":null,"pricing":{"completion":"0.0000006","input_cache_read":"0.000000015","prompt":"0.00000015"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m2-her-20260123","context_length":65536,"created":1769177239,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...","expiration_date":null,"hugging_face_id":"","id":"minimax/minimax-m2-her","knowledge_cutoff":null,"links":{"details":"/api/v1/models/minimax/minimax-m2-her-20260123/endpoints"},"name":"MiniMax: MiniMax M2-her","per_request_limits":null,"pricing":{"completion":"0.0000012","input_cache_read":"0.00000003","prompt":"0.0000003"},"supported_parameters":["max_tokens","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":2048}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"writer/palmyra-x5-20250428","context_length":1040000,"created":1769003823,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...","expiration_date":null,"hugging_face_id":"","id":"writer/palmyra-x5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/writer/palmyra-x5-20250428/endpoints"},"name":"Writer: Palmyra X5","per_request_limits":null,"pricing":{"completion":"0.000006","prompt":"0.0000006"},"supported_parameters":["max_tokens","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1040000,"is_moderated":true,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"liquid/lfm-2.5-1.2b-thinking-20260120","context_length":32768,"created":1768927527,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"LFM2.5-1.2B-Thinking is a lightweight reasoning-focused model optimized for agentic tasks, data extraction, and RAG—while still running comfortably on edge devices. It supports long context (up to 32K tokens) and is...","expiration_date":null,"hugging_face_id":"LiquidAI/LFM2.5-1.2B-Thinking","id":"liquid/lfm-2.5-1.2b-thinking:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/liquid/lfm-2.5-1.2b-thinking-20260120/endpoints"},"name":"LiquidAI: LFM2.5-1.2B-Thinking (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"liquid/lfm-2.5-1.2b-instruct-20260120","context_length":32768,"created":1768927521,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"LFM2.5-1.2B-Instruct is a compact, high-performance instruction-tuned model built for fast on-device AI. It delivers strong chat quality in a 1.2B parameter footprint, with efficient edge inference and broad runtime support.","expiration_date":null,"hugging_face_id":"LiquidAI/LFM2.5-1.2B-Instruct","id":"liquid/lfm-2.5-1.2b-instruct:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/liquid/lfm-2.5-1.2b-instruct-20260120/endpoints"},"name":"LiquidAI: LFM2.5-1.2B-Instruct (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","audio"],"instruct_type":null,"modality":"text+audio->text+audio","output_modalities":["text","audio"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-audio","context_length":128000,"created":1768862569,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-audio","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-audio/endpoints"},"name":"OpenAI: GPT Audio","per_request_limits":null,"pricing":{"audio":"0.000032","completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","audio"],"instruct_type":null,"modality":"text+audio->text+audio","output_modalities":["text","audio"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-audio-mini","context_length":128000,"created":1768859419,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-audio-mini","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-audio-mini/endpoints"},"name":"OpenAI: GPT Audio Mini","per_request_limits":null,"pricing":{"audio":"0.0000006","completion":"0.0000024","prompt":"0.0000006"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.7-flash-20260119","context_length":202752,"created":1768833913,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.7-Flash","id":"z-ai/glm-4.7-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-4.7-flash-20260119/endpoints"},"name":"Z.ai: GLM 4.7 Flash","per_request_limits":null,"pricing":{"completion":"0.0000004","input_cache_read":"0.00000001","prompt":"0.00000006"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":202752,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.2-codex-20260114","context_length":400000,"created":1768409315,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.2-codex","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-codex-20260114/endpoints"},"name":"OpenAI: GPT-5.2-Codex","per_request_limits":null,"pricing":{"completion":"0.000014","input_cache_read":"0.000000175","prompt":"0.00000175"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"bytedance-seed/seed-1.6-flash-20250625","context_length":262144,"created":1766505011,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...","expiration_date":null,"hugging_face_id":"","id":"bytedance-seed/seed-1.6-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-1.6-flash-20250625/endpoints"},"name":"ByteDance Seed: Seed 1.6 Flash","per_request_limits":null,"pricing":{"completion":"0.0000003","prompt":"0.000000075"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"bytedance-seed/seed-1.6-20250625","context_length":262144,"created":1766504997,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.","expiration_date":null,"hugging_face_id":"","id":"bytedance-seed/seed-1.6","knowledge_cutoff":null,"links":{"details":"/api/v1/models/bytedance-seed/seed-1.6-20250625/endpoints"},"name":"ByteDance Seed: Seed 1.6","per_request_limits":null,"pricing":{"completion":"0.000002","prompt":"0.00000025"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m2.1","context_length":196608,"created":1766454997,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.9},"description":"MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...","expiration_date":null,"hugging_face_id":"MiniMaxAI/MiniMax-M2.1","id":"minimax/minimax-m2.1","knowledge_cutoff":null,"links":{"details":"/api/v1/models/minimax/minimax-m2.1/endpoints"},"name":"MiniMax: MiniMax M2.1","per_request_limits":null,"pricing":{"completion":"0.00000095","input_cache_read":"0.00000003","prompt":"0.00000029"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":196608,"is_moderated":false,"max_completion_tokens":196608}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.7-20251222","context_length":202752,"created":1766378014,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.7","id":"z-ai/glm-4.7","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-4.7-20251222/endpoints"},"name":"Z.ai: GLM 4.7","per_request_limits":null,"pricing":{"completion":"0.00000174","prompt":"0.00000038"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":202752,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-3-flash-preview-20251217","context_length":1048576,"created":1765987078,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-3-flash-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemini-3-flash-preview-20251217/endpoints"},"name":"Google: Gemini 3 Flash Preview","per_request_limits":null,"pricing":{"audio":"0.000001","completion":"0.000003","image":"0.0000005","input_cache_read":"0.00000005","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.000003","prompt":"0.0000005","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"xiaomi/mimo-v2-flash-20251210","context_length":262144,"created":1765731308,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":0.95},"description":"MiMo-V2-Flash is an open-source foundation language model developed by Xiaomi. It is a Mixture-of-Experts model with 309B total parameters and 15B active parameters, adopting hybrid attention architecture. MiMo-V2-Flash supports a...","expiration_date":null,"hugging_face_id":"XiaomiMiMo/MiMo-V2-Flash","id":"xiaomi/mimo-v2-flash","knowledge_cutoff":null,"links":{"details":"/api/v1/models/xiaomi/mimo-v2-flash-20251210/endpoints"},"name":"Xiaomi: MiMo-V2-Flash","per_request_limits":null,"pricing":{"completion":"0.00000029","input_cache_read":"0.000000045","prompt":"0.00000009"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","context_length":256000,"created":1765731275,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","id":"nvidia/nemotron-3-nano-30b-a3b:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints"},"name":"NVIDIA: Nemotron 3 Nano 30B A3B (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-3-nano-30b-a3b","context_length":262144,"created":1765731275,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","id":"nvidia/nemotron-3-nano-30b-a3b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nvidia/nemotron-3-nano-30b-a3b/endpoints"},"name":"NVIDIA: Nemotron 3 Nano 30B A3B","per_request_limits":null,"pricing":{"completion":"0.0000002","prompt":"0.00000005"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":228000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.2-chat-20251211","context_length":128000,"created":1765389783,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.2-chat","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-chat-20251211/endpoints"},"name":"OpenAI: GPT-5.2 Chat","per_request_limits":null,"pricing":{"completion":"0.000014","input_cache_read":"0.000000175","prompt":"0.00000175"},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":32000}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.2-pro-20251211","context_length":400000,"created":1765389780,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.2-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-pro-20251211/endpoints"},"name":"OpenAI: GPT-5.2 Pro","per_request_limits":null,"pricing":{"completion":"0.000168","prompt":"0.000021","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.2-20251211","context_length":400000,"created":1765389775,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.2","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.2-20251211/endpoints"},"name":"OpenAI: GPT-5.2","per_request_limits":null,"pricing":{"completion":"0.000014","input_cache_read":"0.000000175","prompt":"0.00000175"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/devstral-2512","context_length":262144,"created":1765285419,"default_parameters":{"frequency_penalty":null,"temperature":0.3,"top_p":null},"description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","expiration_date":null,"hugging_face_id":"mistralai/Devstral-2-123B-Instruct-2512","id":"mistralai/devstral-2512","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/devstral-2512/endpoints"},"name":"Mistral: Devstral 2 2512","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000004","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"relace/relace-search-20251208","context_length":256000,"created":1765213560,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...","expiration_date":null,"hugging_face_id":null,"id":"relace/relace-search","knowledge_cutoff":null,"links":{"details":"/api/v1/models/relace/relace-search-20251208/endpoints"},"name":"Relace: Relace Search","per_request_limits":null,"pricing":{"completion":"0.000003","prompt":"0.000001"},"supported_parameters":["max_tokens","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.6-20251208","context_length":131072,"created":1765207462,"default_parameters":{"frequency_penalty":null,"temperature":0.8,"top_p":0.6},"description":"GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.6V","id":"z-ai/glm-4.6v","knowledge_cutoff":null,"links":{"details":"/api/v1/models/z-ai/glm-4.6-20251208/endpoints"},"name":"Z.ai: GLM 4.6V","per_request_limits":null,"pricing":{"completion":"0.0000009","input_cache_read":"0.00000005","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":24000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"nex-agi/deepseek-v3.1-nex-n1","context_length":131072,"created":1765204393,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"DeepSeek V3.1 Nex-N1 is the flagship release of the Nex-N1 series — a post-trained model designed to highlight agent autonomy, tool use, and real-world productivity. Nex-N1 demonstrates competitive performance across...","expiration_date":null,"hugging_face_id":"nex-agi/DeepSeek-V3.1-Nex-N1","id":"nex-agi/deepseek-v3.1-nex-n1","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nex-agi/deepseek-v3.1-nex-n1/endpoints"},"name":"Nex AGI: DeepSeek V3.1 Nex N1","per_request_limits":null,"pricing":{"completion":"0.0000005","prompt":"0.000000135"},"supported_parameters":["frequency_penalty","max_tokens","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":163840}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"essentialai/rnj-1-instruct","context_length":32768,"created":1765094847,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Rnj-1 is an 8B-parameter, dense, open-weight model family developed by Essential AI and trained from scratch with a focus on programming, math, and scientific reasoning. The model demonstrates strong performance...","expiration_date":null,"hugging_face_id":"EssentialAI/rnj-1-instruct","id":"essentialai/rnj-1-instruct","knowledge_cutoff":null,"links":{"details":"/api/v1/models/essentialai/rnj-1-instruct/endpoints"},"name":"EssentialAI: Rnj 1 Instruct","per_request_limits":null,"pricing":{"completion":"0.00000015","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Router"},"canonical_slug":"openrouter/bodybuilder","context_length":128000,"created":1764903653,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...","expiration_date":null,"hugging_face_id":"","id":"openrouter/bodybuilder","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openrouter/bodybuilder/endpoints"},"name":"Body Builder (beta)","per_request_limits":null,"pricing":{"completion":"-1","prompt":"-1"},"supported_parameters":[],"supported_voices":null,"top_provider":{"context_length":null,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.1-codex-max-20251204","context_length":400000,"created":1764878934,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.1-codex-max","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-codex-max-20251204/endpoints"},"name":"OpenAI: GPT-5.1-Codex-Max","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.000000125","prompt":"0.00000125","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","video","file"],"instruct_type":null,"modality":"text+image+file+video->text","output_modalities":["text"],"tokenizer":"Nova"},"canonical_slug":"amazon/nova-2-lite-v1","context_length":1000000,"created":1764696672,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...","expiration_date":null,"hugging_face_id":"","id":"amazon/nova-2-lite-v1","knowledge_cutoff":null,"links":{"details":"/api/v1/models/amazon/nova-2-lite-v1/endpoints"},"name":"Amazon: Nova 2 Lite","per_request_limits":null,"pricing":{"completion":"0.0000025","prompt":"0.0000003"},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":true,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/ministral-14b-2512","context_length":262144,"created":1764681735,"default_parameters":{"frequency_penalty":null,"temperature":0.3,"top_p":null},"description":"The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...","expiration_date":null,"hugging_face_id":"mistralai/Ministral-3-14B-Instruct-2512","id":"mistralai/ministral-14b-2512","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/ministral-14b-2512/endpoints"},"name":"Mistral: Ministral 3 14B 2512","per_request_limits":null,"pricing":{"completion":"0.0000002","input_cache_read":"0.00000002","prompt":"0.0000002"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/ministral-8b-2512","context_length":262144,"created":1764681654,"default_parameters":{"frequency_penalty":null,"temperature":0.3,"top_p":null},"description":"A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.","expiration_date":null,"hugging_face_id":"mistralai/Ministral-3-8B-Instruct-2512","id":"mistralai/ministral-8b-2512","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/ministral-8b-2512/endpoints"},"name":"Mistral: Ministral 3 8B 2512","per_request_limits":null,"pricing":{"completion":"0.00000015","input_cache_read":"0.000000015","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/ministral-3b-2512","context_length":131072,"created":1764681560,"default_parameters":{"frequency_penalty":null,"temperature":0.3,"top_p":null},"description":"The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.","expiration_date":null,"hugging_face_id":"mistralai/Ministral-3-3B-Instruct-2512","id":"mistralai/ministral-3b-2512","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/ministral-3b-2512/endpoints"},"name":"Mistral: Ministral 3 3B 2512","per_request_limits":null,"pricing":{"completion":"0.0000001","input_cache_read":"0.00000001","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-large-2512","context_length":262144,"created":1764624472,"default_parameters":{"frequency_penalty":null,"temperature":0.0645,"top_p":null},"description":"Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.","expiration_date":null,"hugging_face_id":"","id":"mistralai/mistral-large-2512","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/mistral-large-2512/endpoints"},"name":"Mistral: Mistral Large 3 2512","per_request_limits":null,"pricing":{"completion":"0.0000015","input_cache_read":"0.00000005","prompt":"0.0000005"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/trinity-mini-20251201","context_length":131072,"created":1764601720,"default_parameters":{"frequency_penalty":null,"temperature":0.15,"top_p":0.75},"description":"Trinity Mini is a 26B-parameter (3B active) sparse mixture-of-experts language model featuring 128 experts with 8 active per token. Engineered for efficient reasoning over long contexts (131k) with robust function...","expiration_date":null,"hugging_face_id":"arcee-ai/Trinity-Mini","id":"arcee-ai/trinity-mini","knowledge_cutoff":null,"links":{"details":"/api/v1/models/arcee-ai/trinity-mini-20251201/endpoints"},"name":"Arcee AI: Trinity Mini","per_request_limits":null,"pricing":{"completion":"0.00000015","prompt":"0.000000045"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-v3.2-speciale-20251201","context_length":163840,"created":1764594837,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"DeepSeek-V3.2-Speciale is a high-compute variant of DeepSeek-V3.2 optimized for maximum reasoning and agentic performance. It builds on DeepSeek Sparse Attention (DSA) for efficient long-context processing, then scales post-training reinforcement learning...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3.2-Speciale","id":"deepseek/deepseek-v3.2-speciale","knowledge_cutoff":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-speciale-20251201/endpoints"},"name":"DeepSeek: DeepSeek V3.2 Speciale","per_request_limits":null,"pricing":{"completion":"0.000000431","input_cache_read":"0.000000058","prompt":"0.000000287"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":163840}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-v3.2-20251201","context_length":131072,"created":1764594642,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":0.95},"description":"DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3.2","id":"deepseek/deepseek-v3.2","knowledge_cutoff":null,"links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-20251201/endpoints"},"name":"DeepSeek: DeepSeek V3.2","per_request_limits":null,"pricing":{"completion":"0.000000378","input_cache_read":"0.0000000252","prompt":"0.000000252"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"prime-intellect/intellect-3-20251126","context_length":131072,"created":1764212534,"default_parameters":{"frequency_penalty":null,"temperature":0.6,"top_p":null},"description":"INTELLECT-3 is a 106B-parameter Mixture-of-Experts model (12B active) post-trained from GLM-4.5-Air-Base using supervised fine-tuning (SFT) followed by large-scale reinforcement learning (RL). It offers state-of-the-art performance for its size across math,...","expiration_date":null,"hugging_face_id":"PrimeIntellect/INTELLECT-3-FP8","id":"prime-intellect/intellect-3","knowledge_cutoff":null,"links":{"details":"/api/v1/models/prime-intellect/intellect-3-20251126/endpoints"},"name":"Prime Intellect: INTELLECT-3","per_request_limits":null,"pricing":{"completion":"0.0000011","prompt":"0.0000002"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.5-opus-20251124","context_length":200000,"created":1764010580,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-opus-4.5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-opus-20251124/endpoints"},"name":"Anthropic: Claude Opus 4.5","per_request_limits":null,"pricing":{"completion":"0.000025","input_cache_read":"0.0000005","input_cache_write":"0.00000625","prompt":"0.000005","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","verbosity"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"allenai/olmo-3-32b-think-20251121","context_length":65536,"created":1763758276,"default_parameters":{"frequency_penalty":null,"temperature":0.6,"top_p":0.95},"description":"Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...","expiration_date":null,"hugging_face_id":"allenai/Olmo-3-32B-Think","id":"allenai/olmo-3-32b-think","knowledge_cutoff":null,"links":{"details":"/api/v1/models/allenai/olmo-3-32b-think-20251121/endpoints"},"name":"AllenAI: Olmo 3 32B Think","per_request_limits":null,"pricing":{"completion":"0.0000005","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text+image","output_modalities":["image","text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-3-pro-image-preview-20251120","context_length":65536,"created":1763653797,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-3-pro-image-preview","knowledge_cutoff":null,"links":{"details":"/api/v1/models/google/gemini-3-pro-image-preview-20251120/endpoints"},"name":"Google: Nano Banana Pro (Gemini 3 Pro Image Preview)","per_request_limits":null,"pricing":{"audio":"0.000002","completion":"0.000012","image":"0.000002","input_cache_read":"0.0000002","input_cache_write":"0.000000375","internal_reasoning":"0.000012","prompt":"0.000002","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-4.1-fast","context_length":2000000,"created":1763587502,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.7,"top_k":null,"top_p":0.95},"description":"Grok 4.1 Fast is xAI's best agentic tool calling model that shines in real-world use cases like customer support and deep research. 2M context window. Reasoning can be enabled/disabled using...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-4.1-fast","knowledge_cutoff":null,"links":{"details":"/api/v1/models/x-ai/grok-4.1-fast/endpoints"},"name":"xAI: Grok 4.1 Fast","per_request_limits":null,"pricing":{"completion":"0.0000005","input_cache_read":"0.00000005","prompt":"0.0000002","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":2000000,"is_moderated":false,"max_completion_tokens":30000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"deepcogito/cogito-v2.1-671b-20251118","context_length":128000,"created":1763071233,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...","expiration_date":null,"hugging_face_id":"","id":"deepcogito/cogito-v2.1-671b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/deepcogito/cogito-v2.1-671b-20251118/endpoints"},"name":"Deep Cogito: Cogito v2.1 671B","per_request_limits":null,"pricing":{"completion":"0.00000125","prompt":"0.00000125"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.1-20251113","context_length":400000,"created":1763060305,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.1","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-20251113/endpoints"},"name":"OpenAI: GPT-5.1","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.00000013","prompt":"0.00000125"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.1-chat-20251113","context_length":128000,"created":1763060302,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5.1 Chat (AKA Instant is the fast, lightweight member of the 5.1 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.1-chat","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-chat-20251113/endpoints"},"name":"OpenAI: GPT-5.1 Chat","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.000000125","prompt":"0.00000125","web_search":"0.01"},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.1-codex-20251113","context_length":400000,"created":1763060298,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.1-codex","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-codex-20251113/endpoints"},"name":"OpenAI: GPT-5.1-Codex","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.000000125","prompt":"0.00000125"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5.1-codex-mini-20251113","context_length":400000,"created":1763057820,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5.1-codex-mini","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5.1-codex-mini-20251113/endpoints"},"name":"OpenAI: GPT-5.1-Codex-Mini","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000003","prompt":"0.00000025"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"moonshotai/kimi-k2-thinking-20251106","context_length":262144,"created":1762440622,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...","expiration_date":null,"hugging_face_id":"moonshotai/Kimi-K2-Thinking","id":"moonshotai/kimi-k2-thinking","knowledge_cutoff":null,"links":{"details":"/api/v1/models/moonshotai/kimi-k2-thinking-20251106/endpoints"},"name":"MoonshotAI: Kimi K2 Thinking","per_request_limits":null,"pricing":{"completion":"0.0000025","input_cache_read":"0.00000015","prompt":"0.0000006"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Nova"},"canonical_slug":"amazon/nova-premier-v1","context_length":1000000,"created":1761950332,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.","expiration_date":null,"hugging_face_id":"","id":"amazon/nova-premier-v1","knowledge_cutoff":null,"links":{"details":"/api/v1/models/amazon/nova-premier-v1/endpoints"},"name":"Amazon: Nova Premier 1.0","per_request_limits":null,"pricing":{"completion":"0.0000125","input_cache_read":"0.000000625","prompt":"0.0000025"},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":true,"max_completion_tokens":32000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"perplexity/sonar-pro-search","context_length":200000,"created":1761854366,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...","expiration_date":null,"hugging_face_id":"","id":"perplexity/sonar-pro-search","knowledge_cutoff":null,"links":{"details":"/api/v1/models/perplexity/sonar-pro-search/endpoints"},"name":"Perplexity: Sonar Pro Search","per_request_limits":null,"pricing":{"completion":"0.000015","prompt":"0.000003","web_search":"0.018"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","structured_outputs","temperature","top_k","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":false,"max_completion_tokens":8000}},{"architecture":{"input_modalities":["text","audio"],"instruct_type":null,"modality":"text+audio->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/voxtral-small-24b-2507","context_length":32000,"created":1761835144,"default_parameters":{"frequency_penalty":null,"temperature":0.2,"top_p":0.95},"description":"Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...","expiration_date":null,"hugging_face_id":"mistralai/Voxtral-Small-24B-2507","id":"mistralai/voxtral-small-24b-2507","knowledge_cutoff":null,"links":{"details":"/api/v1/models/mistralai/voxtral-small-24b-2507/endpoints"},"name":"Mistral: Voxtral Small 24B 2507","per_request_limits":null,"pricing":{"audio":"0.0001","completion":"0.0000003","input_cache_read":"0.00000001","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":32000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-oss-safeguard-20b","context_length":131072,"created":1761752836,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...","expiration_date":null,"hugging_face_id":"openai/gpt-oss-safeguard-20b","id":"openai/gpt-oss-safeguard-20b","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-oss-safeguard-20b/endpoints"},"name":"OpenAI: gpt-oss-safeguard-20b","per_request_limits":null,"pricing":{"completion":"0.0000003","input_cache_read":"0.000000037","prompt":"0.000000075"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["image","text","video"],"instruct_type":null,"modality":"text+image+video->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-nano-12b-v2-vl","context_length":128000,"created":1761675565,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16","id":"nvidia/nemotron-nano-12b-v2-vl:free","knowledge_cutoff":null,"links":{"details":"/api/v1/models/nvidia/nemotron-nano-12b-v2-vl/endpoints"},"name":"NVIDIA: Nemotron Nano 12B 2 VL (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m2","context_length":196608,"created":1761252093,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":0.95},"description":"MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...","expiration_date":null,"hugging_face_id":"MiniMaxAI/MiniMax-M2","id":"minimax/minimax-m2","knowledge_cutoff":null,"links":{"details":"/api/v1/models/minimax/minimax-m2/endpoints"},"name":"MiniMax: MiniMax M2","per_request_limits":null,"pricing":{"completion":"0.000001","input_cache_read":"0.00000003","prompt":"0.000000255"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":196608,"is_moderated":false,"max_completion_tokens":196608}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen3-vl-32b-instruct","context_length":131072,"created":1761231332,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1,"temperature":0.7,"top_k":20,"top_p":0.8},"description":"Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-32B-Instruct","id":"qwen/qwen3-vl-32b-instruct","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-32b-instruct/endpoints"},"name":"Qwen: Qwen3 VL 32B Instruct","per_request_limits":null,"pricing":{"completion":"0.000000416","prompt":"0.000000104"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"ibm-granite/granite-4.0-h-micro","context_length":131000,"created":1760927695,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...","expiration_date":null,"hugging_face_id":"ibm-granite/granite-4.0-h-micro","id":"ibm-granite/granite-4.0-h-micro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/ibm-granite/granite-4.0-h-micro/endpoints"},"name":"IBM: Granite 4.0 Micro","per_request_limits":null,"pricing":{"completion":"0.00000011","prompt":"0.000000017"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"microsoft/phi-4-mini-instruct","context_length":128000,"created":1760726049,"default_parameters":{},"description":"Phi-4-mini-instruct is a lightweight open model built upon synthetic data and filtered publicly available websites - with a focus on high-quality, reasoning dense data. The model belongs to the Phi-4...","expiration_date":null,"hugging_face_id":"microsoft/Phi-4-mini-instruct","id":"microsoft/phi-4-mini-instruct","knowledge_cutoff":null,"links":{"details":"/api/v1/models/microsoft/phi-4-mini-instruct/endpoints"},"name":"Microsoft: Phi 4 Mini Instruct","per_request_limits":null,"pricing":{"completion":"0.00000035","input_cache_read":"0.00000008","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text+image","output_modalities":["image","text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-image-mini","context_length":400000,"created":1760624583,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-image-mini","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5-image-mini/endpoints"},"name":"OpenAI: GPT-5 Image Mini","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000025","prompt":"0.0000025","web_search":"0.01"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.5-haiku-20251001","context_length":200000,"created":1760547638,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-haiku-4.5","knowledge_cutoff":null,"links":{"details":"/api/v1/models/anthropic/claude-4.5-haiku-20251001/endpoints"},"name":"Anthropic: Claude Haiku 4.5","per_request_limits":null,"pricing":{"completion":"0.000005","input_cache_read":"0.0000001","input_cache_write":"0.00000125","prompt":"0.000001","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-vl-8b-thinking","context_length":131072,"created":1760463746,"default_parameters":{"temperature":1,"top_p":0.95},"description":"Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-8B-Thinking","id":"qwen/qwen3-vl-8b-thinking","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-8b-thinking/endpoints"},"name":"Qwen: Qwen3 VL 8B Thinking","per_request_limits":null,"pricing":{"completion":"0.000001365","prompt":"0.000000117"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-vl-8b-instruct","context_length":131072,"created":1760463308,"default_parameters":{"frequency_penalty":null,"temperature":0.7,"top_p":0.8},"description":"Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-8B-Instruct","id":"qwen/qwen3-vl-8b-instruct","knowledge_cutoff":null,"links":{"details":"/api/v1/models/qwen/qwen3-vl-8b-instruct/endpoints"},"name":"Qwen: Qwen3 VL 8B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000005","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text+image","output_modalities":["image","text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-image","context_length":400000,"created":1760447986,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-image","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/gpt-5-image/endpoints"},"name":"OpenAI: GPT-5 Image","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.00000125","prompt":"0.00001","web_search":"0.01"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o3-deep-research-2025-06-26","context_length":200000,"created":1760129661,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"o3-deep-research is OpenAI's advanced model for deep research, designed to tackle complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","expiration_date":null,"hugging_face_id":"","id":"openai/o3-deep-research","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/o3-deep-research-2025-06-26/endpoints"},"name":"OpenAI: o3 Deep Research","per_request_limits":null,"pricing":{"completion":"0.00004","input_cache_read":"0.0000025","prompt":"0.00001","web_search":"0.01"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o4-mini-deep-research-2025-06-26","context_length":200000,"created":1760129642,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"o4-mini-deep-research is OpenAI's faster, more affordable deep research model—ideal for tackling complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.","expiration_date":null,"hugging_face_id":"","id":"openai/o4-mini-deep-research","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openai/o4-mini-deep-research-2025-06-26/endpoints"},"name":"OpenAI: o4 Mini Deep Research","per_request_limits":null,"pricing":{"completion":"0.000008","input_cache_read":"0.0000005","prompt":"0.000002","web_search":"0.01"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","presence_penalty","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nvidia/llama-3.3-nemotron-super-49b-v1.5","context_length":131072,"created":1760101395,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":null,"top_p":0.95},"description":"Llama-3.3-Nemotron-Super-49B-v1.5 is a 49B-parameter, English-centric reasoning/chat model derived from Meta’s Llama-3.3-70B-Instruct with a 128K context. It’s post-trained for agentic workflows (RAG, tool calling) via SFT across math, code, science, and...","expiration_date":null,"hugging_face_id":"nvidia/Llama-3_3-Nemotron-Super-49B-v1_5","id":"nvidia/llama-3.3-nemotron-super-49b-v1.5","knowledge_cutoff":"2024-03-31","links":{"details":"/api/v1/models/nvidia/llama-3.3-nemotron-super-49b-v1.5/endpoints"},"name":"NVIDIA: Llama 3.3 Nemotron Super 49B V1.5","per_request_limits":null,"pricing":{"completion":"0.0000004","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/ernie-4.5-21b-a3b-thinking","context_length":131072,"created":1760048887,"default_parameters":{"frequency_penalty":null,"temperature":0.6,"top_p":0.95},"description":"ERNIE-4.5-21B-A3B-Thinking is Baidu's upgraded lightweight MoE model, refined to boost reasoning depth and quality for top-tier performance in logical puzzles, math, science, coding, text generation, and expert-level academic benchmarks.","expiration_date":null,"hugging_face_id":"baidu/ERNIE-4.5-21B-A3B-Thinking","id":"baidu/ernie-4.5-21b-a3b-thinking","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/baidu/ernie-4.5-21b-a3b-thinking/endpoints"},"name":"Baidu: ERNIE 4.5 21B A3B Thinking","per_request_limits":null,"pricing":{"completion":"0.00000028","prompt":"0.00000007"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text+image","output_modalities":["image","text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-flash-image","context_length":32768,"created":1759870431,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-flash-image","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-flash-image/endpoints"},"name":"Google: Nano Banana (Gemini 2.5 Flash Image)","per_request_limits":null,"pricing":{"audio":"0.000001","completion":"0.0000025","image":"0.0000003","input_cache_read":"0.00000003","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.0000025","prompt":"0.0000003","web_search":"0.014"},"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-vl-30b-a3b-thinking","context_length":131072,"created":1759794479,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1,"temperature":0.8,"top_k":20,"top_p":0.95},"description":"Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Thinking","id":"qwen/qwen3-vl-30b-a3b-thinking","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-vl-30b-a3b-thinking/endpoints"},"name":"Qwen: Qwen3 VL 30B A3B Thinking","per_request_limits":null,"pricing":{"completion":"0.00000156","prompt":"0.00000013"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-vl-30b-a3b-instruct","context_length":131072,"created":1759794476,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1,"temperature":0.7,"top_k":20,"top_p":0.8},"description":"Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-30B-A3B-Instruct","id":"qwen/qwen3-vl-30b-a3b-instruct","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-vl-30b-a3b-instruct/endpoints"},"name":"Qwen: Qwen3 VL 30B A3B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000052","prompt":"0.00000013"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-pro-2025-10-06","context_length":400000,"created":1759776663,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-pro","knowledge_cutoff":"2024-09-30","links":{"details":"/api/v1/models/openai/gpt-5-pro-2025-10-06/endpoints"},"name":"OpenAI: GPT-5 Pro","per_request_limits":null,"pricing":{"completion":"0.00012","prompt":"0.000015","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.6","context_length":204800,"created":1759235576,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":null,"top_p":null},"description":"Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.6","id":"z-ai/glm-4.6","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/z-ai/glm-4.6/endpoints"},"name":"Z.ai: GLM 4.6","per_request_limits":null,"pricing":{"completion":"0.0000019","prompt":"0.00000039"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":204800,"is_moderated":false,"max_completion_tokens":204800}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.5-sonnet-20250929","context_length":1000000,"created":1759161676,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":1,"top_k":null,"top_p":1},"description":"Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-sonnet-4.5","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/anthropic/claude-4.5-sonnet-20250929/endpoints"},"name":"Anthropic: Claude Sonnet 4.5","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.0000003","input_cache_write":"0.00000375","prompt":"0.000003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":true,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-v3.1","modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-v3.2-exp","context_length":163840,"created":1759150481,"default_parameters":{"frequency_penalty":null,"temperature":0.6,"top_p":0.95},"description":"DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3.2-Exp","id":"deepseek/deepseek-v3.2-exp","knowledge_cutoff":"2025-07-31","links":{"details":"/api/v1/models/deepseek/deepseek-v3.2-exp/endpoints"},"name":"DeepSeek: DeepSeek V3.2 Exp","per_request_limits":null,"pricing":{"completion":"0.00000041","prompt":"0.00000027"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"thedrummer/cydonia-24b-v4.1","context_length":131072,"created":1758931878,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.","expiration_date":null,"hugging_face_id":"thedrummer/cydonia-24b-v4.1","id":"thedrummer/cydonia-24b-v4.1","knowledge_cutoff":"2024-04-30","links":{"details":"/api/v1/models/thedrummer/cydonia-24b-v4.1/endpoints"},"name":"TheDrummer: Cydonia 24B V4.1","per_request_limits":null,"pricing":{"completion":"0.0000005","input_cache_read":"0.00000015","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"relace/relace-apply-3","context_length":256000,"created":1758891572,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...","expiration_date":null,"hugging_face_id":"","id":"relace/relace-apply-3","knowledge_cutoff":null,"links":{"details":"/api/v1/models/relace/relace-apply-3/endpoints"},"name":"Relace: Relace Apply 3","per_request_limits":null,"pricing":{"completion":"0.00000125","prompt":"0.00000085"},"supported_parameters":["max_tokens","seed","stop"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-flash-lite-preview-09-2025","context_length":1048576,"created":1758819686,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-flash-lite-preview-09-2025","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-flash-lite-preview-09-2025/endpoints"},"name":"Google: Gemini 2.5 Flash Lite Preview 09-2025","per_request_limits":null,"pricing":{"audio":"0.0000003","completion":"0.0000004","image":"0.0000001","input_cache_read":"0.00000001","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.0000004","prompt":"0.0000001","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-vl-235b-a22b-thinking","context_length":131072,"created":1758668690,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":1,"temperature":0.8,"top_k":20,"top_p":0.95},"description":"Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Thinking","id":"qwen/qwen3-vl-235b-a22b-thinking","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-vl-235b-a22b-thinking/endpoints"},"name":"Qwen: Qwen3 VL 235B A22B Thinking","per_request_limits":null,"pricing":{"completion":"0.0000026","prompt":"0.00000026"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-vl-235b-a22b-instruct","context_length":262144,"created":1758668687,"default_parameters":{"frequency_penalty":null,"temperature":0.7,"top_p":0.8},"description":"Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-VL-235B-A22B-Instruct","id":"qwen/qwen3-vl-235b-a22b-instruct","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-vl-235b-a22b-instruct/endpoints"},"name":"Qwen: Qwen3 VL 235B A22B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000088","input_cache_read":"0.00000011","prompt":"0.0000002"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-max","context_length":262144,"created":1758662808,"default_parameters":{"frequency_penalty":null,"temperature":1,"top_p":1},"description":"Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen3-max","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-max/endpoints"},"name":"Qwen: Qwen3 Max","per_request_limits":null,"pricing":{"completion":"0.0000039","input_cache_read":"0.000000156","input_cache_write":"0.000000975","prompt":"0.00000078"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-coder-plus","context_length":1000000,"created":1758662707,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen3-coder-plus","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-coder-plus/endpoints"},"name":"Qwen: Qwen3 Coder Plus","per_request_limits":null,"pricing":{"completion":"0.00000325","input_cache_read":"0.00000013","input_cache_write":"0.0000008125","prompt":"0.00000065"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-codex","context_length":400000,"created":1758643403,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-codex","knowledge_cutoff":"2024-09-30","links":{"details":"/api/v1/models/openai/gpt-5-codex/endpoints"},"name":"OpenAI: GPT-5 Codex","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.000000125","prompt":"0.00000125"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-v3.1","modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-v3.1-terminus","context_length":163840,"created":1758548275,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3.1-Terminus","id":"deepseek/deepseek-v3.1-terminus","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/deepseek/deepseek-v3.1-terminus/endpoints"},"name":"DeepSeek: DeepSeek V3.1 Terminus","per_request_limits":null,"pricing":{"completion":"0.00000095","input_cache_read":"0.00000013","prompt":"0.00000027"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-4-fast","context_length":2000000,"created":1758240090,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Grok 4 Fast is xAI's latest multimodal model with SOTA cost-efficiency and a 2M token context window. It comes in two flavors: non-reasoning and reasoning. Read more about the model...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-4-fast","knowledge_cutoff":"2025-09-30","links":{"details":"/api/v1/models/x-ai/grok-4-fast/endpoints"},"name":"xAI: Grok 4 Fast","per_request_limits":null,"pricing":{"completion":"0.0000005","input_cache_read":"0.00000005","prompt":"0.0000002","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":2000000,"is_moderated":false,"max_completion_tokens":30000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"alibaba/tongyi-deepresearch-30b-a3b","context_length":131072,"created":1758210804,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Tongyi DeepResearch is an agentic large language model developed by Tongyi Lab, with 30 billion total parameters activating only 3 billion per token. It's optimized for long-horizon, deep information-seeking tasks...","expiration_date":null,"hugging_face_id":"Alibaba-NLP/Tongyi-DeepResearch-30B-A3B","id":"alibaba/tongyi-deepresearch-30b-a3b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/alibaba/tongyi-deepresearch-30b-a3b/endpoints"},"name":"Tongyi DeepResearch 30B A3B","per_request_limits":null,"pricing":{"completion":"0.00000045","input_cache_read":"0.00000009","prompt":"0.00000009"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-coder-flash","context_length":1000000,"created":1758115536,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen3-coder-flash","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-coder-flash/endpoints"},"name":"Qwen: Qwen3 Coder Flash","per_request_limits":null,"pricing":{"completion":"0.000000975","input_cache_read":"0.000000039","input_cache_write":"0.00000024375","prompt":"0.000000195"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-next-80b-a3b-thinking-2509","context_length":131072,"created":1757612284,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Thinking","id":"qwen/qwen3-next-80b-a3b-thinking","knowledge_cutoff":"2025-09-30","links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-thinking-2509/endpoints"},"name":"Qwen: Qwen3 Next 80B A3B Thinking","per_request_limits":null,"pricing":{"completion":"0.00000078","prompt":"0.0000000975"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","context_length":262144,"created":1757612213,"default_parameters":{},"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Instruct","id":"qwen/qwen3-next-80b-a3b-instruct:free","knowledge_cutoff":"2025-09-30","links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-instruct-2509/endpoints"},"name":"Qwen: Qwen3 Next 80B A3B Instruct (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-next-80b-a3b-instruct-2509","context_length":262144,"created":1757612213,"default_parameters":{},"description":"Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Next-80B-A3B-Instruct","id":"qwen/qwen3-next-80b-a3b-instruct","knowledge_cutoff":"2025-09-30","links":{"details":"/api/v1/models/qwen/qwen3-next-80b-a3b-instruct-2509/endpoints"},"name":"Qwen: Qwen3 Next 80B A3B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000011","prompt":"0.00000009"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen-plus-2025-07-28","context_length":1000000,"created":1757347599,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-plus-2025-07-28:thinking","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-plus-2025-07-28/endpoints"},"name":"Qwen: Qwen Plus 0728 (thinking)","per_request_limits":null,"pricing":{"completion":"0.00000078","input_cache_write":"0.000000325","prompt":"0.00000026"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen-plus-2025-07-28","context_length":1000000,"created":1757347599,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-plus-2025-07-28","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-plus-2025-07-28/endpoints"},"name":"Qwen: Qwen Plus 0728","per_request_limits":null,"pricing":{"completion":"0.00000078","input_cache_write":"0.000000325","prompt":"0.00000026"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-nano-9b-v2","context_length":128000,"created":1757106807,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","id":"nvidia/nemotron-nano-9b-v2:free","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/nvidia/nemotron-nano-9b-v2/endpoints"},"name":"NVIDIA: Nemotron Nano 9B V2 (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nvidia/nemotron-nano-9b-v2","context_length":131072,"created":1757106807,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...","expiration_date":null,"hugging_face_id":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","id":"nvidia/nemotron-nano-9b-v2","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/nvidia/nemotron-nano-9b-v2/endpoints"},"name":"NVIDIA: Nemotron Nano 9B V2","per_request_limits":null,"pricing":{"completion":"0.00000016","prompt":"0.00000004"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"moonshotai/kimi-k2-0905","context_length":262144,"created":1757021147,"default_parameters":{},"description":"Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...","expiration_date":null,"hugging_face_id":"moonshotai/Kimi-K2-Instruct-0905","id":"moonshotai/kimi-k2-0905","knowledge_cutoff":"2024-12-31","links":{"details":"/api/v1/models/moonshotai/kimi-k2-0905/endpoints"},"name":"MoonshotAI: Kimi K2 0905","per_request_limits":null,"pricing":{"completion":"0.000002","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-30b-a3b-thinking-2507","context_length":131072,"created":1756399192,"default_parameters":{},"description":"Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-30B-A3B-Thinking-2507","id":"qwen/qwen3-30b-a3b-thinking-2507","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-thinking-2507/endpoints"},"name":"Qwen: Qwen3 30B A3B Thinking 2507","per_request_limits":null,"pricing":{"completion":"0.0000004","input_cache_read":"0.00000008","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-code-fast-1","context_length":256000,"created":1756238927,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Grok Code Fast 1 is a speedy and economical reasoning model that excels at agentic coding. With reasoning traces visible in the response, developers can steer Grok Code for high-quality...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-code-fast-1","knowledge_cutoff":"2025-09-30","links":{"details":"/api/v1/models/x-ai/grok-code-fast-1/endpoints"},"name":"xAI: Grok Code Fast 1","per_request_limits":null,"pricing":{"completion":"0.0000015","input_cache_read":"0.00000002","prompt":"0.0000002","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":10000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nousresearch/hermes-4-70b","context_length":131072,"created":1756236182,"default_parameters":{},"description":"Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...","expiration_date":null,"hugging_face_id":"NousResearch/Hermes-4-70B","id":"nousresearch/hermes-4-70b","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/nousresearch/hermes-4-70b/endpoints"},"name":"Nous: Hermes 4 70B","per_request_limits":null,"pricing":{"completion":"0.0000004","prompt":"0.00000013"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"nousresearch/hermes-4-405b","context_length":131072,"created":1756235463,"default_parameters":{},"description":"Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...","expiration_date":null,"hugging_face_id":"NousResearch/Hermes-4-405B","id":"nousresearch/hermes-4-405b","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/nousresearch/hermes-4-405b/endpoints"},"name":"Nous: Hermes 4 405B","per_request_limits":null,"pricing":{"completion":"0.000003","prompt":"0.000001"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-v3.1","modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-chat-v3.1","context_length":32768,"created":1755779628,"default_parameters":{},"description":"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3.1","id":"deepseek/deepseek-chat-v3.1","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3.1/endpoints"},"name":"DeepSeek: DeepSeek V3.1","per_request_limits":null,"pricing":{"completion":"0.00000075","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":7168}},{"architecture":{"input_modalities":["audio","text"],"instruct_type":null,"modality":"text+audio->text+audio","output_modalities":["text","audio"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-audio-preview","context_length":128000,"created":1755233061,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The gpt-4o-audio-preview model adds support for audio inputs as prompts. This enhancement allows the model to detect nuances within audio recordings and add depth to generated user experiences. Audio outputs...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4o-audio-preview","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-audio-preview/endpoints"},"name":"OpenAI: GPT-4o Audio","per_request_limits":null,"pricing":{"audio":"0.00004","completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-medium-3.1","context_length":131072,"created":1755095639,"default_parameters":{"temperature":0.3},"description":"Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...","expiration_date":null,"hugging_face_id":"","id":"mistralai/mistral-medium-3.1","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/mistralai/mistral-medium-3.1/endpoints"},"name":"Mistral: Mistral Medium 3.1","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000004","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/ernie-4.5-21b-a3b","context_length":120000,"created":1755034167,"default_parameters":{"frequency_penalty":null,"temperature":0.8,"top_p":0.8},"description":"A sophisticated text-based Mixture-of-Experts (MoE) model featuring 21B total parameters with 3B activated per token, delivering exceptional multimodal understanding and generation through heterogeneous MoE structures and modality-isolated routing. Supporting an...","expiration_date":null,"hugging_face_id":"baidu/ERNIE-4.5-21B-A3B-PT","id":"baidu/ernie-4.5-21b-a3b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/baidu/ernie-4.5-21b-a3b/endpoints"},"name":"Baidu: ERNIE 4.5 21B A3B","per_request_limits":null,"pricing":{"completion":"0.00000028","prompt":"0.00000007"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":120000,"is_moderated":false,"max_completion_tokens":8000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/ernie-4.5-vl-28b-a3b","context_length":30000,"created":1755032836,"default_parameters":{},"description":"A powerful multimodal Mixture-of-Experts chat model featuring 28B total parameters with 3B activated per token, delivering exceptional text and vision understanding through its innovative heterogeneous MoE structure with modality-isolated routing....","expiration_date":null,"hugging_face_id":"baidu/ERNIE-4.5-VL-28B-A3B-PT","id":"baidu/ernie-4.5-vl-28b-a3b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/baidu/ernie-4.5-vl-28b-a3b/endpoints"},"name":"Baidu: ERNIE 4.5 VL 28B A3B","per_request_limits":null,"pricing":{"completion":"0.00000056","prompt":"0.00000014"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":30000,"is_moderated":false,"max_completion_tokens":8000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.5v","context_length":65536,"created":1754922288,"default_parameters":{"frequency_penalty":null,"temperature":0.75,"top_p":null},"description":"GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.5V","id":"z-ai/glm-4.5v","knowledge_cutoff":"2024-12-31","links":{"details":"/api/v1/models/z-ai/glm-4.5v/endpoints"},"name":"Z.ai: GLM 4.5V","per_request_limits":null,"pricing":{"completion":"0.0000018","input_cache_read":"0.00000011","prompt":"0.0000006"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"ai21/jamba-large-1.7","context_length":256000,"created":1754669020,"default_parameters":{},"description":"Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...","expiration_date":null,"hugging_face_id":"ai21labs/AI21-Jamba-Large-1.7","id":"ai21/jamba-large-1.7","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/ai21/jamba-large-1.7/endpoints"},"name":"AI21: Jamba Large 1.7","per_request_limits":null,"pricing":{"completion":"0.000008","prompt":"0.000002"},"supported_parameters":["max_tokens","response_format","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["file","image","text"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-chat-2025-08-07","context_length":128000,"created":1754587837,"default_parameters":{},"description":"GPT-5 Chat is designed for advanced, natural, multimodal, and context-aware conversations for enterprise applications.","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-chat","knowledge_cutoff":"2024-09-30","links":{"details":"/api/v1/models/openai/gpt-5-chat-2025-08-07/endpoints"},"name":"OpenAI: GPT-5 Chat","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.000000125","prompt":"0.00000125","web_search":"0.01"},"supported_parameters":["max_tokens","response_format","seed","structured_outputs"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-2025-08-07","context_length":400000,"created":1754587413,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5","knowledge_cutoff":"2024-09-30","links":{"details":"/api/v1/models/openai/gpt-5-2025-08-07/endpoints"},"name":"OpenAI: GPT-5","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.000000125","prompt":"0.00000125"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-mini-2025-08-07","context_length":400000,"created":1754587407,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-mini","knowledge_cutoff":"2024-05-31","links":{"details":"/api/v1/models/openai/gpt-5-mini-2025-08-07/endpoints"},"name":"OpenAI: GPT-5 Mini","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.000000025","prompt":"0.00000025","web_search":"0.01"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":true,"max_completion_tokens":128000}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-5-nano-2025-08-07","context_length":400000,"created":1754587402,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-5-nano","knowledge_cutoff":"2024-05-31","links":{"details":"/api/v1/models/openai/gpt-5-nano-2025-08-07/endpoints"},"name":"OpenAI: GPT-5 Nano","per_request_limits":null,"pricing":{"completion":"0.0000004","input_cache_read":"0.00000001","prompt":"0.00000005"},"supported_parameters":["include_reasoning","max_completion_tokens","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":400000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-oss-120b","context_length":131072,"created":1754414231,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","expiration_date":null,"hugging_face_id":"openai/gpt-oss-120b","id":"openai/gpt-oss-120b:free","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"},"name":"OpenAI: gpt-oss-120b (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","stop","temperature","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":true,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-oss-120b","context_length":131072,"created":1754414231,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...","expiration_date":null,"hugging_face_id":"openai/gpt-oss-120b","id":"openai/gpt-oss-120b","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-oss-120b/endpoints"},"name":"OpenAI: gpt-oss-120b","per_request_limits":null,"pricing":{"completion":"0.00000018","prompt":"0.000000039"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-oss-20b","context_length":131072,"created":1754414229,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","expiration_date":null,"hugging_face_id":"openai/gpt-oss-20b","id":"openai/gpt-oss-20b:free","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"},"name":"OpenAI: gpt-oss-20b (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","stop","temperature","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":true,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-oss-20b","context_length":131072,"created":1754414229,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...","expiration_date":null,"hugging_face_id":"openai/gpt-oss-20b","id":"openai/gpt-oss-20b","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-oss-20b/endpoints"},"name":"OpenAI: gpt-oss-20b","per_request_limits":null,"pricing":{"completion":"0.00000014","prompt":"0.00000003"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4.1-opus-20250805","context_length":200000,"created":1754411591,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-opus-4.1","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/anthropic/claude-4.1-opus-20250805/endpoints"},"name":"Anthropic: Claude Opus 4.1","per_request_limits":null,"pricing":{"completion":"0.000075","input_cache_read":"0.0000015","input_cache_write":"0.00001875","prompt":"0.000015","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":false,"max_completion_tokens":32000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/codestral-2508","context_length":256000,"created":1754079630,"default_parameters":{"temperature":0.3},"description":"Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)","expiration_date":null,"hugging_face_id":"","id":"mistralai/codestral-2508","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/mistralai/codestral-2508/endpoints"},"name":"Mistral: Codestral 2508","per_request_limits":null,"pricing":{"completion":"0.0000009","input_cache_read":"0.00000003","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-coder-30b-a3b-instruct","context_length":160000,"created":1753972379,"default_parameters":{},"description":"Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","id":"qwen/qwen3-coder-30b-a3b-instruct","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-coder-30b-a3b-instruct/endpoints"},"name":"Qwen: Qwen3 Coder 30B A3B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000027","prompt":"0.00000007"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":160000,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-30b-a3b-instruct-2507","context_length":262144,"created":1753806965,"default_parameters":{},"description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-30B-A3B-Instruct-2507","id":"qwen/qwen3-30b-a3b-instruct-2507","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-instruct-2507/endpoints"},"name":"Qwen: Qwen3 30B A3B Instruct 2507","per_request_limits":null,"pricing":{"completion":"0.0000003","prompt":"0.00000009"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":262144}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.5","context_length":131072,"created":1753471347,"default_parameters":{"frequency_penalty":null,"temperature":0.75,"top_p":null},"description":"GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.5","id":"z-ai/glm-4.5","knowledge_cutoff":"2024-12-31","links":{"details":"/api/v1/models/z-ai/glm-4.5/endpoints"},"name":"Z.ai: GLM 4.5","per_request_limits":null,"pricing":{"completion":"0.0000022","input_cache_read":"0.00000011","prompt":"0.0000006"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":98304}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.5-air","context_length":131072,"created":1753471258,"default_parameters":{"frequency_penalty":null,"temperature":0.75,"top_p":null},"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.5-Air","id":"z-ai/glm-4.5-air:free","knowledge_cutoff":"2024-12-31","links":{"details":"/api/v1/models/z-ai/glm-4.5-air/endpoints"},"name":"Z.ai: GLM 4.5 Air (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":96000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4.5-air","context_length":131072,"created":1753471258,"default_parameters":{"frequency_penalty":null,"temperature":0.75,"top_p":null},"description":"GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...","expiration_date":null,"hugging_face_id":"zai-org/GLM-4.5-Air","id":"z-ai/glm-4.5-air","knowledge_cutoff":"2024-12-31","links":{"details":"/api/v1/models/z-ai/glm-4.5-air/endpoints"},"name":"Z.ai: GLM 4.5 Air","per_request_limits":null,"pricing":{"completion":"0.00000085","input_cache_read":"0.000000025","prompt":"0.00000013"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":98304}},{"architecture":{"input_modalities":["text"],"instruct_type":"qwen3","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-235b-a22b-thinking-2507","context_length":131072,"created":1753449557,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-235B-A22B-Thinking-2507","id":"qwen/qwen3-235b-a22b-thinking-2507","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-thinking-2507/endpoints"},"name":"Qwen: Qwen3 235B A22B Thinking 2507","per_request_limits":null,"pricing":{"completion":"0.000001495","prompt":"0.0000001495"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"z-ai/glm-4-32b-0414","context_length":128000,"created":1753376617,"default_parameters":{"frequency_penalty":null,"temperature":0.75,"top_p":null},"description":"GLM 4 32B is a cost-effective foundation language model. It can efficiently perform complex tasks and has significantly enhanced capabilities in tool use, online search, and code-related intelligent tasks. It...","expiration_date":null,"hugging_face_id":"","id":"z-ai/glm-4-32b","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/z-ai/glm-4-32b-0414/endpoints"},"name":"Z.ai: GLM 4 32B ","per_request_limits":null,"pricing":{"completion":"0.0000001","prompt":"0.0000001"},"supported_parameters":["max_tokens","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","context_length":262000,"created":1753230546,"default_parameters":{},"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","id":"qwen/qwen3-coder:free","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-coder-480b-a35b-07-25/endpoints"},"name":"Qwen: Qwen3 Coder 480B A35B (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262000,"is_moderated":false,"max_completion_tokens":262000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-coder-480b-a35b-07-25","context_length":262144,"created":1753230546,"default_parameters":{},"description":"Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","id":"qwen/qwen3-coder","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-coder-480b-a35b-07-25/endpoints"},"name":"Qwen: Qwen3 Coder 480B A35B","per_request_limits":null,"pricing":{"completion":"0.0000018","prompt":"0.00000022"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"bytedance/ui-tars-1.5-7b","context_length":128000,"created":1753205056,"default_parameters":{},"description":"UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...","expiration_date":null,"hugging_face_id":"ByteDance-Seed/UI-TARS-1.5-7B","id":"bytedance/ui-tars-1.5-7b","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/bytedance/ui-tars-1.5-7b/endpoints"},"name":"ByteDance: UI-TARS 7B ","per_request_limits":null,"pricing":{"completion":"0.0000002","input_cache_read":"0.0000001","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":2048}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-flash-lite","context_length":1048576,"created":1753200276,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-flash-lite","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-flash-lite/endpoints"},"name":"Google: Gemini 2.5 Flash Lite","per_request_limits":null,"pricing":{"audio":"0.0000003","completion":"0.0000004","image":"0.0000001","input_cache_read":"0.00000001","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.0000004","prompt":"0.0000001","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-235b-a22b-07-25","context_length":262144,"created":1753119555,"default_parameters":{},"description":"Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-235B-A22B-Instruct-2507","id":"qwen/qwen3-235b-a22b-2507","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-07-25/endpoints"},"name":"Qwen: Qwen3 235B A22B Instruct 2507","per_request_limits":null,"pricing":{"completion":"0.0000001","prompt":"0.000000071"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"switchpoint/router","context_length":131072,"created":1752272899,"default_parameters":{},"description":"Switchpoint AI's router instantly analyzes your request and directs it to the optimal AI from an ever-evolving library. As the world of LLMs advances, our router gets smarter, ensuring you...","expiration_date":null,"hugging_face_id":"","id":"switchpoint/router","knowledge_cutoff":null,"links":{"details":"/api/v1/models/switchpoint/router/endpoints"},"name":"Switchpoint Router","per_request_limits":null,"pricing":{"completion":"0.0000034","prompt":"0.00000085"},"supported_parameters":["include_reasoning","max_tokens","reasoning","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"moonshotai/kimi-k2","context_length":131072,"created":1752263252,"default_parameters":{},"description":"Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...","expiration_date":null,"hugging_face_id":"moonshotai/Kimi-K2-Instruct","id":"moonshotai/kimi-k2","knowledge_cutoff":"2024-12-31","links":{"details":"/api/v1/models/moonshotai/kimi-k2/endpoints"},"name":"MoonshotAI: Kimi K2 0711","per_request_limits":null,"pricing":{"completion":"0.0000023","prompt":"0.00000057"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/devstral-medium-2507","context_length":131072,"created":1752161321,"default_parameters":{"temperature":0.3},"description":"Devstral Medium is a high-performance code generation and agentic reasoning model developed jointly by Mistral AI and All Hands AI. Positioned as a step up from Devstral Small, it achieves...","expiration_date":null,"hugging_face_id":"","id":"mistralai/devstral-medium","knowledge_cutoff":"2025-06-30","links":{"details":"/api/v1/models/mistralai/devstral-medium-2507/endpoints"},"name":"Mistral: Devstral Medium","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000004","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/devstral-small-2507","context_length":131072,"created":1752160751,"default_parameters":{"temperature":0.3},"description":"Devstral Small 1.1 is a 24B parameter open-weight language model for software engineering agents, developed by Mistral AI in collaboration with All Hands AI. Finetuned from Mistral Small 3.1 and...","expiration_date":null,"hugging_face_id":"mistralai/Devstral-Small-2507","id":"mistralai/devstral-small","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/mistralai/devstral-small-2507/endpoints"},"name":"Mistral: Devstral Small 1.1","per_request_limits":null,"pricing":{"completion":"0.0000003","input_cache_read":"0.00000001","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"venice/uncensored","context_length":32768,"created":1752094966,"default_parameters":{},"description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","expiration_date":null,"hugging_face_id":"cognitivecomputations/Dolphin-Mistral-24B-Venice-Edition","id":"cognitivecomputations/dolphin-mistral-24b-venice-edition:free","knowledge_cutoff":"2024-04-30","links":{"details":"/api/v1/models/venice/uncensored/endpoints"},"name":"Venice: Uncensored (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-4-07-09","context_length":256000,"created":1752087689,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Grok 4 is xAI's latest reasoning model with a 256k context window. It supports parallel tool calling, structured outputs, and both image and text inputs. Note that reasoning is not...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-4","knowledge_cutoff":"2025-07-31","links":{"details":"/api/v1/models/x-ai/grok-4-07-09/endpoints"},"name":"xAI: Grok 4","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.00000075","prompt":"0.000003","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"tencent/hunyuan-a13b-instruct","context_length":131072,"created":1751987664,"default_parameters":{},"description":"Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...","expiration_date":null,"hugging_face_id":"tencent/Hunyuan-A13B-Instruct","id":"tencent/hunyuan-a13b-instruct","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/tencent/hunyuan-a13b-instruct/endpoints"},"name":"Tencent: Hunyuan A13B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000057","prompt":"0.00000014"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","reasoning","response_format","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"tngtech/deepseek-r1t2-chimera","context_length":163840,"created":1751986985,"default_parameters":{},"description":"DeepSeek-TNG-R1T2-Chimera is the second-generation Chimera model from TNG Tech. It is a 671 B-parameter mixture-of-experts text-generation model assembled from DeepSeek-AI’s R1-0528, R1, and V3-0324 checkpoints with an Assembly-of-Experts merge. The...","expiration_date":null,"hugging_face_id":"tngtech/DeepSeek-TNG-R1T2-Chimera","id":"tngtech/deepseek-r1t2-chimera","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/tngtech/deepseek-r1t2-chimera/endpoints"},"name":"TNG: DeepSeek R1T2 Chimera","per_request_limits":null,"pricing":{"completion":"0.0000011","input_cache_read":"0.00000015","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":163840}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"morph/morph-v3-large","context_length":262144,"created":1751910858,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code>...","expiration_date":null,"hugging_face_id":"","id":"morph/morph-v3-large","knowledge_cutoff":null,"links":{"details":"/api/v1/models/morph/morph-v3-large/endpoints"},"name":"Morph: Morph V3 Large","per_request_limits":null,"pricing":{"completion":"0.0000019","prompt":"0.0000009"},"supported_parameters":["max_tokens","stop","temperature"],"supported_voices":null,"top_provider":{"context_length":262144,"is_moderated":false,"max_completion_tokens":131072}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"morph/morph-v3-fast","context_length":81920,"created":1751910002,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: <instruction>{instruction}</instruction> <code>{initial_code}</code> <update>{edit_snippet}</update>...","expiration_date":null,"hugging_face_id":"","id":"morph/morph-v3-fast","knowledge_cutoff":null,"links":{"details":"/api/v1/models/morph/morph-v3-fast/endpoints"},"name":"Morph: Morph V3 Fast","per_request_limits":null,"pricing":{"completion":"0.0000012","prompt":"0.0000008"},"supported_parameters":["max_tokens","stop","temperature"],"supported_voices":null,"top_provider":{"context_length":81920,"is_moderated":false,"max_completion_tokens":38000}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/ernie-4.5-vl-424b-a47b","context_length":123000,"created":1751300903,"default_parameters":{},"description":"ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...","expiration_date":null,"hugging_face_id":"baidu/ERNIE-4.5-VL-424B-A47B-PT","id":"baidu/ernie-4.5-vl-424b-a47b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/baidu/ernie-4.5-vl-424b-a47b/endpoints"},"name":"Baidu: ERNIE 4.5 VL 424B A47B ","per_request_limits":null,"pricing":{"completion":"0.00000125","prompt":"0.00000042"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":123000,"is_moderated":false,"max_completion_tokens":16000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"baidu/ernie-4.5-300b-a47b","context_length":123000,"created":1751300139,"default_parameters":{},"description":"ERNIE-4.5-300B-A47B is a 300B parameter Mixture-of-Experts (MoE) language model developed by Baidu as part of the ERNIE 4.5 series. It activates 47B parameters per token and supports text generation in...","expiration_date":null,"hugging_face_id":"baidu/ERNIE-4.5-300B-A47B-PT","id":"baidu/ernie-4.5-300b-a47b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/baidu/ernie-4.5-300b-a47b/endpoints"},"name":"Baidu: ERNIE 4.5 300B A47B ","per_request_limits":null,"pricing":{"completion":"0.0000011","prompt":"0.00000028"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":123000,"is_moderated":false,"max_completion_tokens":12000}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-small-3.2-24b-instruct-2506","context_length":128000,"created":1750443016,"default_parameters":{"temperature":0.3},"description":"Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...","expiration_date":null,"hugging_face_id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","id":"mistralai/mistral-small-3.2-24b-instruct","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/mistralai/mistral-small-3.2-24b-instruct-2506/endpoints"},"name":"Mistral: Mistral Small 3.2 24B","per_request_limits":null,"pricing":{"completion":"0.0000002","prompt":"0.000000075"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-m1","context_length":1000000,"created":1750200414,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...","expiration_date":null,"hugging_face_id":"","id":"minimax/minimax-m1","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/minimax/minimax-m1/endpoints"},"name":"MiniMax: MiniMax M1","per_request_limits":null,"pricing":{"completion":"0.0000022","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":40000}},{"architecture":{"input_modalities":["file","image","text","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-flash","context_length":1048576,"created":1750172488,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-flash","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-flash/endpoints"},"name":"Google: Gemini 2.5 Flash","per_request_limits":null,"pricing":{"audio":"0.000001","completion":"0.0000025","image":"0.0000003","input_cache_read":"0.00000003","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.0000025","prompt":"0.0000003","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-pro","context_length":1048576,"created":1750169544,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-pro","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-pro/endpoints"},"name":"Google: Gemini 2.5 Pro","per_request_limits":null,"pricing":{"audio":"0.00000125","completion":"0.00001","image":"0.00000125","input_cache_read":"0.000000125","input_cache_write":"0.000000375","internal_reasoning":"0.00001","prompt":"0.00000125","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","file","image"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o3-pro-2025-06-10","context_length":200000,"created":1749598352,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...","expiration_date":null,"hugging_face_id":"","id":"openai/o3-pro","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/o3-pro-2025-06-10/endpoints"},"name":"OpenAI: o3 Pro","per_request_limits":null,"pricing":{"completion":"0.00008","prompt":"0.00002","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-3-mini","context_length":131072,"created":1749583245,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-3-mini","knowledge_cutoff":"2025-02-28","links":{"details":"/api/v1/models/x-ai/grok-3-mini/endpoints"},"name":"xAI: Grok 3 Mini","per_request_limits":null,"pricing":{"completion":"0.0000005","input_cache_read":"0.000000075","prompt":"0.0000003","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-3","context_length":131072,"created":1749582908,"default_parameters":{},"description":"Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-3","knowledge_cutoff":"2025-02-28","links":{"details":"/api/v1/models/x-ai/grok-3/endpoints"},"name":"xAI: Grok 3","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.00000075","prompt":"0.000003","web_search":"0.005"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["file","image","text","audio"],"instruct_type":null,"modality":"text+image+file+audio->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-pro-preview-06-05","context_length":1048576,"created":1749137257,"default_parameters":{},"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-pro-preview","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-pro-preview-06-05/endpoints"},"name":"Google: Gemini 2.5 Pro Preview 06-05","per_request_limits":null,"pricing":{"audio":"0.00000125","completion":"0.00001","image":"0.00000125","input_cache_read":"0.000000125","input_cache_write":"0.000000375","internal_reasoning":"0.00001","prompt":"0.00000125","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-r1","modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-r1-0528","context_length":163840,"created":1748455170,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-R1-0528","id":"deepseek/deepseek-r1-0528","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/deepseek/deepseek-r1-0528/endpoints"},"name":"DeepSeek: R1 0528","per_request_limits":null,"pricing":{"completion":"0.00000215","input_cache_read":"0.00000035","prompt":"0.0000005"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4-opus-20250522","context_length":200000,"created":1747931245,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-opus-4","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/anthropic/claude-4-opus-20250522/endpoints"},"name":"Anthropic: Claude Opus 4","per_request_limits":null,"pricing":{"completion":"0.000075","input_cache_read":"0.0000015","input_cache_write":"0.00001875","prompt":"0.000015","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":false,"max_completion_tokens":32000}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-4-sonnet-20250522","context_length":1000000,"created":1747930371,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...","expiration_date":null,"hugging_face_id":"","id":"anthropic/claude-sonnet-4","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/anthropic/claude-4-sonnet-20250522/endpoints"},"name":"Anthropic: Claude Sonnet 4","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.0000003","input_cache_write":"0.00000375","prompt":"0.000003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"google/gemma-3n-e4b-it","context_length":32768,"created":1747776824,"default_parameters":{},"description":"Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...","expiration_date":null,"hugging_face_id":"google/gemma-3n-E4B-it","id":"google/gemma-3n-e4b-it","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/google/gemma-3n-e4b-it/endpoints"},"name":"Google: Gemma 3n 4B","per_request_limits":null,"pricing":{"completion":"0.00000012","prompt":"0.00000006"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-medium-3","context_length":131072,"created":1746627341,"default_parameters":{"temperature":0.3},"description":"Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...","expiration_date":null,"hugging_face_id":"","id":"mistralai/mistral-medium-3","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/mistralai/mistral-medium-3/endpoints"},"name":"Mistral: Mistral Medium 3","per_request_limits":null,"pricing":{"completion":"0.000002","input_cache_read":"0.00000004","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.5-pro-preview-03-25","context_length":1048576,"created":1746578513,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...","expiration_date":null,"hugging_face_id":"","id":"google/gemini-2.5-pro-preview-05-06","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/google/gemini-2.5-pro-preview-03-25/endpoints"},"name":"Google: Gemini 2.5 Pro Preview 05-06","per_request_limits":null,"pricing":{"audio":"0.00000125","completion":"0.00001","image":"0.00000125","input_cache_read":"0.000000125","input_cache_write":"0.000000375","internal_reasoning":"0.00001","prompt":"0.00000125","web_search":"0.014"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":65535}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/spotlight","context_length":131072,"created":1746481552,"default_parameters":{},"description":"Spotlight is a 7‑billion‑parameter vision‑language model derived from Qwen 2.5‑VL and fine‑tuned by Arcee AI for tight image‑text grounding tasks. It offers a 32 k‑token context window, enabling rich multimodal...","expiration_date":null,"hugging_face_id":"","id":"arcee-ai/spotlight","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/arcee-ai/spotlight/endpoints"},"name":"Arcee AI: Spotlight","per_request_limits":null,"pricing":{"completion":"0.00000018","prompt":"0.00000018"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":65537}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/maestro-reasoning","context_length":131072,"created":1746481269,"default_parameters":{},"description":"Maestro Reasoning is Arcee's flagship analysis model: a 32 B‑parameter derivative of Qwen 2.5‑32 B tuned with DPO and chain‑of‑thought RL for step‑by‑step logic. Compared to the earlier 7 B...","expiration_date":null,"hugging_face_id":"","id":"arcee-ai/maestro-reasoning","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/arcee-ai/maestro-reasoning/endpoints"},"name":"Arcee AI: Maestro Reasoning","per_request_limits":null,"pricing":{"completion":"0.0000033","prompt":"0.0000009"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/virtuoso-large","context_length":131072,"created":1746478885,"default_parameters":{},"description":"Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...","expiration_date":null,"hugging_face_id":"","id":"arcee-ai/virtuoso-large","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/arcee-ai/virtuoso-large/endpoints"},"name":"Arcee AI: Virtuoso Large","per_request_limits":null,"pricing":{"completion":"0.0000012","prompt":"0.00000075"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"arcee-ai/coder-large","context_length":32768,"created":1746478663,"default_parameters":{},"description":"Coder‑Large is a 32 B‑parameter offspring of Qwen 2.5‑Instruct that has been further trained on permissively‑licensed GitHub, CodeSearchNet and synthetic bug‑fix corpora. It supports a 32k context window, enabling multi‑file...","expiration_date":null,"hugging_face_id":"","id":"arcee-ai/coder-large","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/arcee-ai/coder-large/endpoints"},"name":"Arcee AI: Coder Large","per_request_limits":null,"pricing":{"completion":"0.0000008","prompt":"0.0000005"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["image","text"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"meta-llama/llama-guard-4-12b","context_length":163840,"created":1745975193,"default_parameters":{},"description":"Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-Guard-4-12B","id":"meta-llama/llama-guard-4-12b","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/meta-llama/llama-guard-4-12b/endpoints"},"name":"Meta: Llama Guard 4 12B","per_request_limits":null,"pricing":{"completion":"0.00000018","prompt":"0.00000018"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"qwen3","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-30b-a3b-04-28","context_length":40960,"created":1745878604,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-30B-A3B","id":"qwen/qwen3-30b-a3b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-30b-a3b-04-28/endpoints"},"name":"Qwen: Qwen3 30B A3B","per_request_limits":null,"pricing":{"completion":"0.00000045","prompt":"0.00000009"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":40960,"is_moderated":false,"max_completion_tokens":20000}},{"architecture":{"input_modalities":["text"],"instruct_type":"qwen3","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-8b-04-28","context_length":40960,"created":1745876632,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":0.6,"top_k":20,"top_p":0.95},"description":"Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-8B","id":"qwen/qwen3-8b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-8b-04-28/endpoints"},"name":"Qwen: Qwen3 8B","per_request_limits":null,"pricing":{"completion":"0.0000004","input_cache_read":"0.00000005","prompt":"0.00000005"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":40960,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":"qwen3","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-14b-04-28","context_length":40960,"created":1745876478,"default_parameters":{},"description":"Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-14B","id":"qwen/qwen3-14b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-14b-04-28/endpoints"},"name":"Qwen: Qwen3 14B","per_request_limits":null,"pricing":{"completion":"0.00000024","prompt":"0.00000006"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":40960,"is_moderated":false,"max_completion_tokens":40960}},{"architecture":{"input_modalities":["text"],"instruct_type":"qwen3","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-32b-04-28","context_length":40960,"created":1745875945,"default_parameters":{},"description":"Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-32B","id":"qwen/qwen3-32b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-32b-04-28/endpoints"},"name":"Qwen: Qwen3 32B","per_request_limits":null,"pricing":{"completion":"0.00000024","input_cache_read":"0.00000004","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":40960,"is_moderated":false,"max_completion_tokens":40960}},{"architecture":{"input_modalities":["text"],"instruct_type":"qwen3","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen3"},"canonical_slug":"qwen/qwen3-235b-a22b-04-28","context_length":131072,"created":1745875757,"default_parameters":{},"description":"Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...","expiration_date":null,"hugging_face_id":"Qwen/Qwen3-235B-A22B","id":"qwen/qwen3-235b-a22b","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen3-235b-a22b-04-28/endpoints"},"name":"Qwen: Qwen3 235B A22B","per_request_limits":null,"pricing":{"completion":"0.00000182","prompt":"0.000000455"},"supported_parameters":["include_reasoning","max_tokens","presence_penalty","reasoning","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o4-mini-high-2025-04-16","context_length":200000,"created":1744824212,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...","expiration_date":null,"hugging_face_id":"","id":"openai/o4-mini-high","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/o4-mini-high-2025-04-16/endpoints"},"name":"OpenAI: o4 Mini High","per_request_limits":null,"pricing":{"completion":"0.0000044","input_cache_read":"0.000000275","prompt":"0.0000011","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o3-2025-04-16","context_length":200000,"created":1744823457,"default_parameters":{},"description":"o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....","expiration_date":null,"hugging_face_id":"","id":"openai/o3","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/o3-2025-04-16/endpoints"},"name":"OpenAI: o3","per_request_limits":null,"pricing":{"completion":"0.000008","input_cache_read":"0.0000005","prompt":"0.000002","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o4-mini-2025-04-16","context_length":200000,"created":1744820942,"default_parameters":{},"description":"OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...","expiration_date":null,"hugging_face_id":"","id":"openai/o4-mini","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/o4-mini-2025-04-16/endpoints"},"name":"OpenAI: o4 Mini","per_request_limits":null,"pricing":{"completion":"0.0000044","input_cache_read":"0.000000275","prompt":"0.0000011","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4.1-2025-04-14","context_length":1047576,"created":1744651385,"default_parameters":{},"description":"GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4.1","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-4.1-2025-04-14/endpoints"},"name":"OpenAI: GPT-4.1","per_request_limits":null,"pricing":{"completion":"0.000008","input_cache_read":"0.0000005","prompt":"0.000002"},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1047576,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4.1-mini-2025-04-14","context_length":1047576,"created":1744651381,"default_parameters":{},"description":"GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4.1-mini","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-4.1-mini-2025-04-14/endpoints"},"name":"OpenAI: GPT-4.1 Mini","per_request_limits":null,"pricing":{"completion":"0.0000016","input_cache_read":"0.0000001","prompt":"0.0000004","web_search":"0.01"},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1047576,"is_moderated":true,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["image","text","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4.1-nano-2025-04-14","context_length":1047576,"created":1744651369,"default_parameters":{},"description":"For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4.1-nano","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/openai/gpt-4.1-nano-2025-04-14/endpoints"},"name":"OpenAI: GPT-4.1 Nano","per_request_limits":null,"pricing":{"completion":"0.0000004","input_cache_read":"0.000000025","prompt":"0.0000001","web_search":"0.01"},"supported_parameters":["max_completion_tokens","max_tokens","response_format","seed","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1047576,"is_moderated":true,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":"alpaca","modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"alfredpros/codellama-7b-instruct-solidity","context_length":4096,"created":1744641874,"default_parameters":{},"description":"A finetuned 7 billion parameters Code LLaMA - Instruct model to generate Solidity smart contract using 4-bit QLoRA finetuning provided by PEFT library.","expiration_date":null,"hugging_face_id":"AlfredPros/CodeLlama-7b-Instruct-Solidity","id":"alfredpros/codellama-7b-instruct-solidity","knowledge_cutoff":"2023-06-30","links":{"details":"/api/v1/models/alfredpros/codellama-7b-instruct-solidity/endpoints"},"name":"AlfredPros: CodeLLaMa 7B Instruct Solidity","per_request_limits":null,"pricing":{"completion":"0.0000012","prompt":"0.0000008"},"supported_parameters":["frequency_penalty","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":4096,"is_moderated":false,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-3-mini-beta","context_length":131072,"created":1744240195,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Grok 3 Mini is a lightweight, smaller thinking model. Unlike traditional models that generate answers immediately, Grok 3 Mini thinks before responding. It’s ideal for reasoning-heavy tasks that don’t demand...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-3-mini-beta","knowledge_cutoff":"2025-02-28","links":{"details":"/api/v1/models/x-ai/grok-3-mini-beta/endpoints"},"name":"xAI: Grok 3 Mini Beta","per_request_limits":null,"pricing":{"completion":"0.0000005","input_cache_read":"0.000000075","prompt":"0.0000003","web_search":"0.005"},"supported_parameters":["include_reasoning","logprobs","max_tokens","reasoning","response_format","seed","stop","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Grok"},"canonical_slug":"x-ai/grok-3-beta","context_length":131072,"created":1744240068,"default_parameters":{},"description":"Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...","expiration_date":"2026-05-15","hugging_face_id":"","id":"x-ai/grok-3-beta","knowledge_cutoff":"2025-02-28","links":{"details":"/api/v1/models/x-ai/grok-3-beta/endpoints"},"name":"xAI: Grok 3 Beta","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.00000075","prompt":"0.000003","web_search":"0.005"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Llama4"},"canonical_slug":"meta-llama/llama-4-maverick-17b-128e-instruct","context_length":1048576,"created":1743881822,"default_parameters":{},"description":"Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct","id":"meta-llama/llama-4-maverick","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/meta-llama/llama-4-maverick-17b-128e-instruct/endpoints"},"name":"Meta: Llama 4 Maverick","per_request_limits":null,"pricing":{"completion":"0.0000006","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Llama4"},"canonical_slug":"meta-llama/llama-4-scout-17b-16e-instruct","context_length":327680,"created":1743881519,"default_parameters":{},"description":"Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","id":"meta-llama/llama-4-scout","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/meta-llama/llama-4-scout-17b-16e-instruct/endpoints"},"name":"Meta: Llama 4 Scout","per_request_limits":null,"pricing":{"completion":"0.0000003","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":327680,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-chat-v3-0324","context_length":163840,"created":1742824755,"default_parameters":{},"description":"DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3-0324","id":"deepseek/deepseek-chat-v3-0324","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3-0324/endpoints"},"name":"DeepSeek: DeepSeek V3 0324","per_request_limits":null,"pricing":{"completion":"0.00000077","input_cache_read":"0.000000135","prompt":"0.0000002"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o1-pro","context_length":200000,"created":1742423211,"default_parameters":{},"description":"The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...","expiration_date":null,"hugging_face_id":"","id":"openai/o1-pro","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/o1-pro/endpoints"},"name":"OpenAI: o1-pro","per_request_limits":null,"pricing":{"completion":"0.0006","prompt":"0.00015"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-small-3.1-24b-instruct-2503","context_length":128000,"created":1742238937,"default_parameters":{"temperature":0.3},"description":"Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...","expiration_date":null,"hugging_face_id":"mistralai/Mistral-Small-3.1-24B-Instruct-2503","id":"mistralai/mistral-small-3.1-24b-instruct","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/mistralai/mistral-small-3.1-24b-instruct-2503/endpoints"},"name":"Mistral: Mistral Small 3.1 24B","per_request_limits":null,"pricing":{"completion":"0.00000056","prompt":"0.00000035"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":"gemma","modality":"text+image->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemma-3-4b-it","context_length":131072,"created":1741905510,"default_parameters":{},"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","expiration_date":null,"hugging_face_id":"google/gemma-3-4b-it","id":"google/gemma-3-4b-it","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/google/gemma-3-4b-it/endpoints"},"name":"Google: Gemma 3 4B","per_request_limits":null,"pricing":{"completion":"0.00000008","prompt":"0.00000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":"gemma","modality":"text+image->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemma-3-12b-it","context_length":131072,"created":1741902625,"default_parameters":{},"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","expiration_date":null,"hugging_face_id":"google/gemma-3-12b-it","id":"google/gemma-3-12b-it","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/google/gemma-3-12b-it/endpoints"},"name":"Google: Gemma 3 12B","per_request_limits":null,"pricing":{"completion":"0.00000013","prompt":"0.00000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"cohere/command-a-03-2025","context_length":256000,"created":1741894342,"default_parameters":{},"description":"Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...","expiration_date":null,"hugging_face_id":"CohereForAI/c4ai-command-a-03-2025","id":"cohere/command-a","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/cohere/command-a-03-2025/endpoints"},"name":"Cohere: Command A","per_request_limits":null,"pricing":{"completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":256000,"is_moderated":true,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-mini-search-preview-2025-03-11","context_length":128000,"created":1741818122,"default_parameters":{},"description":"GPT-4o mini Search Preview is a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4o-mini-search-preview","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-mini-search-preview-2025-03-11/endpoints"},"name":"OpenAI: GPT-4o-mini Search Preview","per_request_limits":null,"pricing":{"completion":"0.0000006","prompt":"0.00000015","web_search":"0.0275"},"supported_parameters":["max_tokens","response_format","structured_outputs","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-search-preview-2025-03-11","context_length":128000,"created":1741817949,"default_parameters":{},"description":"GPT-4o Search Previewis a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4o-search-preview","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-search-preview-2025-03-11/endpoints"},"name":"OpenAI: GPT-4o Search Preview","per_request_limits":null,"pricing":{"completion":"0.00001","prompt":"0.0000025","web_search":"0.035"},"supported_parameters":["max_tokens","response_format","structured_outputs","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"rekaai/reka-flash-3","context_length":65536,"created":1741812813,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...","expiration_date":null,"hugging_face_id":"RekaAI/reka-flash-3","id":"rekaai/reka-flash-3","knowledge_cutoff":"2025-01-31","links":{"details":"/api/v1/models/rekaai/reka-flash-3/endpoints"},"name":"Reka Flash 3","per_request_limits":null,"pricing":{"completion":"0.0000002","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":65536}},{"architecture":{"input_modalities":["text","image"],"instruct_type":"gemma","modality":"text+image->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemma-3-27b-it","context_length":131072,"created":1741756359,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...","expiration_date":null,"hugging_face_id":"google/gemma-3-27b-it","id":"google/gemma-3-27b-it","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/google/gemma-3-27b-it/endpoints"},"name":"Google: Gemma 3 27B","per_request_limits":null,"pricing":{"completion":"0.00000016","prompt":"0.00000008"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"thedrummer/skyfall-36b-v2","context_length":32768,"created":1741636566,"default_parameters":{},"description":"Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.","expiration_date":null,"hugging_face_id":"TheDrummer/Skyfall-36B-v2","id":"thedrummer/skyfall-36b-v2","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/thedrummer/skyfall-36b-v2/endpoints"},"name":"TheDrummer: Skyfall 36B V2","per_request_limits":null,"pricing":{"completion":"0.0000008","input_cache_read":"0.00000025","prompt":"0.00000055"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":"deepseek-r1","modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"perplexity/sonar-reasoning-pro","context_length":128000,"created":1741313308,"default_parameters":{},"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...","expiration_date":null,"hugging_face_id":"","id":"perplexity/sonar-reasoning-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/perplexity/sonar-reasoning-pro/endpoints"},"name":"Perplexity: Sonar Reasoning Pro","per_request_limits":null,"pricing":{"completion":"0.000008","prompt":"0.000002","web_search":"0.005"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"perplexity/sonar-pro","context_length":200000,"created":1741312423,"default_parameters":{},"description":"Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...","expiration_date":null,"hugging_face_id":"","id":"perplexity/sonar-pro","knowledge_cutoff":null,"links":{"details":"/api/v1/models/perplexity/sonar-pro/endpoints"},"name":"Perplexity: Sonar Pro","per_request_limits":null,"pricing":{"completion":"0.000015","prompt":"0.000003","web_search":"0.005"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":false,"max_completion_tokens":8000}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-r1","modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"perplexity/sonar-deep-research","context_length":128000,"created":1741311246,"default_parameters":{},"description":"Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...","expiration_date":null,"hugging_face_id":"","id":"perplexity/sonar-deep-research","knowledge_cutoff":null,"links":{"details":"/api/v1/models/perplexity/sonar-deep-research/endpoints"},"name":"Perplexity: Sonar Deep Research","per_request_limits":null,"pricing":{"completion":"0.000008","internal_reasoning":"0.000003","prompt":"0.000002","web_search":"0.005"},"supported_parameters":["frequency_penalty","include_reasoning","max_tokens","presence_penalty","reasoning","temperature","top_k","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.0-flash-lite-001","context_length":1048576,"created":1740506212,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini 2.0 Flash Lite offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5),...","expiration_date":"2026-06-01","hugging_face_id":"","id":"google/gemini-2.0-flash-lite-001","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/google/gemini-2.0-flash-lite-001/endpoints"},"name":"Google: Gemini 2.0 Flash Lite","per_request_limits":null,"pricing":{"audio":"0.000000075","completion":"0.0000003","image":"0.000000075","internal_reasoning":"0.0000003","prompt":"0.000000075","web_search":"0.014"},"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-3-7-sonnet-20250219","context_length":200000,"created":1740422110,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and...","expiration_date":"2026-05-11","hugging_face_id":"","id":"anthropic/claude-3.7-sonnet","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/anthropic/claude-3-7-sonnet-20250219/endpoints"},"name":"Anthropic: Claude 3.7 Sonnet","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.0000003","input_cache_write":"0.00000375","prompt":"0.000003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":false,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-3-7-sonnet-20250219","context_length":200000,"created":1740422110,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and...","expiration_date":"2026-05-11","hugging_face_id":"","id":"anthropic/claude-3.7-sonnet:thinking","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/anthropic/claude-3-7-sonnet-20250219/endpoints"},"name":"Anthropic: Claude 3.7 Sonnet (thinking)","per_request_limits":null,"pricing":{"completion":"0.000015","input_cache_read":"0.0000003","input_cache_write":"0.00000375","prompt":"0.000003","web_search":"0.01"},"supported_parameters":["include_reasoning","max_tokens","reasoning","stop","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":false,"max_completion_tokens":64000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-saba-2502","context_length":32768,"created":1739803239,"default_parameters":{"temperature":0.3},"description":"Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...","expiration_date":null,"hugging_face_id":"","id":"mistralai/mistral-saba","knowledge_cutoff":"2024-09-30","links":{"details":"/api/v1/models/mistralai/mistral-saba-2502/endpoints"},"name":"Mistral: Saba","per_request_limits":null,"pricing":{"completion":"0.0000006","input_cache_read":"0.00000002","prompt":"0.0000002"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"none","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-guard-3-8b","context_length":131072,"created":1739401318,"default_parameters":{},"description":"Llama Guard 3 is a Llama-3.1-8B pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM inputs (prompt classification)...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-Guard-3-8B","id":"meta-llama/llama-guard-3-8b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-guard-3-8b/endpoints"},"name":"Llama Guard 3 8B","per_request_limits":null,"pricing":{"completion":"0.00000003","prompt":"0.00000048"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","file"],"instruct_type":null,"modality":"text+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o3-mini-high-2025-01-31","context_length":200000,"created":1739372611,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...","expiration_date":null,"hugging_face_id":"","id":"openai/o3-mini-high","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/o3-mini-high-2025-01-31/endpoints"},"name":"OpenAI: o3 Mini High","per_request_limits":null,"pricing":{"completion":"0.0000044","input_cache_read":"0.00000055","prompt":"0.0000011"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["text","image","file","audio","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemini-2.0-flash-001","context_length":1048576,"created":1738769413,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Gemini Flash 2.0 offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5). It...","expiration_date":"2026-06-01","hugging_face_id":"","id":"google/gemini-2.0-flash-001","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/google/gemini-2.0-flash-001/endpoints"},"name":"Google: Gemini 2.0 Flash","per_request_limits":null,"pricing":{"audio":"0.0000007","completion":"0.0000004","image":"0.0000001","input_cache_read":"0.000000025","input_cache_write":"0.00000008333333333333334","internal_reasoning":"0.0000004","prompt":"0.0000001","web_search":"0.014"},"supported_parameters":["max_tokens","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1048576,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-vl-plus","context_length":131072,"created":1738731255,"default_parameters":{},"description":"Qwen's Enhanced Large Visual Language Model. Significantly upgraded for detailed recognition capabilities and text recognition abilities, supporting ultra-high pixel resolutions up to millions of pixels and extreme aspect ratios for...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-vl-plus","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-vl-plus/endpoints"},"name":"Qwen: Qwen VL Plus","per_request_limits":null,"pricing":{"completion":"0.0000004095","input_cache_read":"0.0000000273","prompt":"0.0000001365"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"aion-labs/aion-1.0","context_length":131072,"created":1738697557,"default_parameters":{},"description":"Aion-1.0 is a multi-model system designed for high performance across various tasks, including reasoning and coding. It is built on DeepSeek-R1, augmented with additional models and techniques such as Tree...","expiration_date":null,"hugging_face_id":"","id":"aion-labs/aion-1.0","knowledge_cutoff":null,"links":{"details":"/api/v1/models/aion-labs/aion-1.0/endpoints"},"name":"AionLabs: Aion-1.0","per_request_limits":null,"pricing":{"completion":"0.000008","prompt":"0.000004"},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"aion-labs/aion-1.0-mini","context_length":131072,"created":1738697107,"default_parameters":{},"description":"Aion-1.0-Mini 32B parameter model is a distilled version of the DeepSeek-R1 model, designed for strong performance in reasoning domains such as mathematics, coding, and logic. It is a modified variant...","expiration_date":null,"hugging_face_id":"FuseAI/FuseO1-DeepSeekR1-QwQ-SkyT1-32B-Preview","id":"aion-labs/aion-1.0-mini","knowledge_cutoff":null,"links":{"details":"/api/v1/models/aion-labs/aion-1.0-mini/endpoints"},"name":"AionLabs: Aion-1.0-Mini","per_request_limits":null,"pricing":{"completion":"0.0000014","prompt":"0.0000007"},"supported_parameters":["include_reasoning","max_tokens","reasoning","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"aion-labs/aion-rp-llama-3.1-8b","context_length":32768,"created":1738696718,"default_parameters":{},"description":"Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...","expiration_date":null,"hugging_face_id":"","id":"aion-labs/aion-rp-llama-3.1-8b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/aion-labs/aion-rp-llama-3.1-8b/endpoints"},"name":"AionLabs: Aion-RP 1.0 (8B)","per_request_limits":null,"pricing":{"completion":"0.0000016","prompt":"0.0000008"},"supported_parameters":["max_tokens","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-vl-max-2025-01-25","context_length":131072,"created":1738434304,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen VL Max is a visual understanding model with 7500 tokens context length. It excels in delivering optimal performance for a broader spectrum of complex tasks.\n","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-vl-max","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-vl-max-2025-01-25/endpoints"},"name":"Qwen: Qwen VL Max","per_request_limits":null,"pricing":{"completion":"0.00000208","prompt":"0.00000052"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-turbo-2024-11-01","context_length":131072,"created":1738410974,"default_parameters":{},"description":"Qwen-Turbo, based on Qwen2.5, is a 1M context model that provides fast speed and low cost, suitable for simple tasks.","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-turbo","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-turbo-2024-11-01/endpoints"},"name":"Qwen: Qwen-Turbo","per_request_limits":null,"pricing":{"completion":"0.00000013","input_cache_read":"0.0000000065","prompt":"0.0000000325"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen2.5-vl-72b-instruct","context_length":32000,"created":1738410311,"default_parameters":{},"description":"Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.","expiration_date":null,"hugging_face_id":"Qwen/Qwen2.5-VL-72B-Instruct","id":"qwen/qwen2.5-vl-72b-instruct","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/qwen/qwen2.5-vl-72b-instruct/endpoints"},"name":"Qwen: Qwen2.5 VL 72B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000075","prompt":"0.00000025"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-plus-2025-01-25","context_length":1000000,"created":1738409840,"default_parameters":{},"description":"Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-plus","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-plus-2025-01-25/endpoints"},"name":"Qwen: Qwen-Plus","per_request_limits":null,"pricing":{"completion":"0.00000078","input_cache_read":"0.000000052","input_cache_write":"0.000000325","prompt":"0.00000026"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":1000000,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-max-2025-01-25","context_length":32768,"created":1738402289,"default_parameters":{},"description":"Qwen-Max, based on Qwen2.5, provides the best inference performance among [Qwen models](/qwen), especially for complex multi-step tasks. It's a large-scale MoE model that has been pretrained on over 20 trillion...","expiration_date":null,"hugging_face_id":"","id":"qwen/qwen-max","knowledge_cutoff":"2025-03-31","links":{"details":"/api/v1/models/qwen/qwen-max-2025-01-25/endpoints"},"name":"Qwen: Qwen-Max ","per_request_limits":null,"pricing":{"completion":"0.00000416","input_cache_read":"0.000000208","prompt":"0.00000104"},"supported_parameters":["max_tokens","presence_penalty","response_format","seed","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text","file"],"instruct_type":null,"modality":"text+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o3-mini-2025-01-31","context_length":200000,"created":1738351721,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...","expiration_date":null,"hugging_face_id":"","id":"openai/o3-mini","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/o3-mini-2025-01-31/endpoints"},"name":"OpenAI: o3 Mini","per_request_limits":null,"pricing":{"completion":"0.0000044","input_cache_read":"0.00000055","prompt":"0.0000011"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-small-24b-instruct-2501","context_length":32768,"created":1738255409,"default_parameters":{"frequency_penalty":null,"temperature":0.3,"top_p":null},"description":"Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...","expiration_date":null,"hugging_face_id":"mistralai/Mistral-Small-24B-Instruct-2501","id":"mistralai/mistral-small-24b-instruct-2501","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/mistralai/mistral-small-24b-instruct-2501/endpoints"},"name":"Mistral: Mistral Small 3","per_request_limits":null,"pricing":{"completion":"0.00000008","prompt":"0.00000005"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-r1","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"deepseek/deepseek-r1-distill-qwen-32b","context_length":32768,"created":1738194830,"default_parameters":{},"description":"DeepSeek R1 Distill Qwen 32B is a distilled large language model based on [Qwen 2.5 32B](https://huggingface.co/Qwen/Qwen2.5-32B), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). It outperforms OpenAI's o1-mini across various benchmarks, achieving new...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-R1-Distill-Qwen-32B","id":"deepseek/deepseek-r1-distill-qwen-32b","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/deepseek/deepseek-r1-distill-qwen-32b/endpoints"},"name":"DeepSeek: R1 Distill Qwen 32B","per_request_limits":null,"pricing":{"completion":"0.00000029","prompt":"0.00000029"},"supported_parameters":["frequency_penalty","include_reasoning","logprobs","max_tokens","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"perplexity/sonar","context_length":127072,"created":1738013808,"default_parameters":{},"description":"Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...","expiration_date":null,"hugging_face_id":"","id":"perplexity/sonar","knowledge_cutoff":null,"links":{"details":"/api/v1/models/perplexity/sonar/endpoints"},"name":"Perplexity: Sonar","per_request_limits":null,"pricing":{"completion":"0.000001","prompt":"0.000001","web_search":"0.005"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","temperature","top_k","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":127072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-r1","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"deepseek/deepseek-r1-distill-llama-70b","context_length":131072,"created":1737663169,"default_parameters":{},"description":"DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-R1-Distill-Llama-70B","id":"deepseek/deepseek-r1-distill-llama-70b","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/deepseek/deepseek-r1-distill-llama-70b/endpoints"},"name":"DeepSeek: R1 Distill Llama 70B","per_request_limits":null,"pricing":{"completion":"0.0000008","prompt":"0.0000007"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"deepseek-r1","modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-r1","context_length":64000,"created":1737381095,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-R1","id":"deepseek/deepseek-r1","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/deepseek/deepseek-r1/endpoints"},"name":"DeepSeek: R1","per_request_limits":null,"pricing":{"completion":"0.0000025","prompt":"0.0000007"},"supported_parameters":["frequency_penalty","include_reasoning","max_completion_tokens","max_tokens","presence_penalty","reasoning","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":64000,"is_moderated":false,"max_completion_tokens":16000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"minimax/minimax-01","context_length":1000192,"created":1736915462,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...","expiration_date":null,"hugging_face_id":"MiniMaxAI/MiniMax-Text-01","id":"minimax/minimax-01","knowledge_cutoff":"2024-03-31","links":{"details":"/api/v1/models/minimax/minimax-01/endpoints"},"name":"MiniMax: MiniMax-01","per_request_limits":null,"pricing":{"completion":"0.0000011","prompt":"0.0000002"},"supported_parameters":["max_tokens","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":1000192,"is_moderated":false,"max_completion_tokens":1000192}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"microsoft/phi-4","context_length":16384,"created":1736489872,"default_parameters":{},"description":"[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...","expiration_date":null,"hugging_face_id":"microsoft/phi-4","id":"microsoft/phi-4","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/microsoft/phi-4/endpoints"},"name":"Microsoft: Phi 4","per_request_limits":null,"pricing":{"completion":"0.00000014","prompt":"0.000000065"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":16384,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"sao10k/l3.1-70b-hanami-x1","context_length":16000,"created":1736302854,"default_parameters":{},"description":"This is [Sao10K](/sao10k)'s experiment over [Euryale v2.2](/sao10k/l3.1-euryale-70b).","expiration_date":null,"hugging_face_id":"Sao10K/L3.1-70B-Hanami-x1","id":"sao10k/l3.1-70b-hanami-x1","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/sao10k/l3.1-70b-hanami-x1/endpoints"},"name":"Sao10K: Llama 3.1 70B Hanami x1","per_request_limits":null,"pricing":{"completion":"0.000003","prompt":"0.000003"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":16000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"DeepSeek"},"canonical_slug":"deepseek/deepseek-chat-v3","context_length":163840,"created":1735241320,"default_parameters":{},"description":"DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...","expiration_date":null,"hugging_face_id":"deepseek-ai/DeepSeek-V3","id":"deepseek/deepseek-chat","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/deepseek/deepseek-chat-v3/endpoints"},"name":"DeepSeek: DeepSeek V3","per_request_limits":null,"pricing":{"completion":"0.00000089","prompt":"0.00000032"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":163840,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"sao10k/l3.3-euryale-70b-v2.3","context_length":131072,"created":1734535928,"default_parameters":{},"description":"Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).","expiration_date":null,"hugging_face_id":"Sao10K/L3.3-70B-Euryale-v2.3","id":"sao10k/l3.3-euryale-70b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/sao10k/l3.3-euryale-70b-v2.3/endpoints"},"name":"Sao10K: Llama 3.3 Euryale 70B","per_request_limits":null,"pricing":{"completion":"0.00000075","prompt":"0.00000065"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/o1-2024-12-17","context_length":200000,"created":1734459999,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...","expiration_date":null,"hugging_face_id":"","id":"openai/o1","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/o1-2024-12-17/endpoints"},"name":"OpenAI: o1","per_request_limits":null,"pricing":{"completion":"0.00006","input_cache_read":"0.0000075","prompt":"0.000015"},"supported_parameters":["include_reasoning","max_tokens","reasoning","response_format","seed","structured_outputs","tool_choice","tools"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":100000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Cohere"},"canonical_slug":"cohere/command-r7b-12-2024","context_length":128000,"created":1734158152,"default_parameters":{},"description":"Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...","expiration_date":null,"hugging_face_id":"","id":"cohere/command-r7b-12-2024","knowledge_cutoff":"2024-08-31","links":{"details":"/api/v1/models/cohere/command-r7b-12-2024/endpoints"},"name":"Cohere: Command R7B (12-2024)","per_request_limits":null,"pricing":{"completion":"0.00000015","prompt":"0.0000000375"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":4000}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.3-70b-instruct","context_length":65536,"created":1733506137,"default_parameters":{},"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","id":"meta-llama/llama-3.3-70b-instruct:free","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.3-70b-instruct/endpoints"},"name":"Meta: Llama 3.3 70B Instruct (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.3-70b-instruct","context_length":131072,"created":1733506137,"default_parameters":{},"description":"The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-3.3-70B-Instruct","id":"meta-llama/llama-3.3-70b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.3-70b-instruct/endpoints"},"name":"Meta: Llama 3.3 70B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000032","prompt":"0.0000001"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Nova"},"canonical_slug":"amazon/nova-lite-v1","context_length":300000,"created":1733437363,"default_parameters":{},"description":"Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...","expiration_date":null,"hugging_face_id":"","id":"amazon/nova-lite-v1","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/amazon/nova-lite-v1/endpoints"},"name":"Amazon: Nova Lite 1.0","per_request_limits":null,"pricing":{"completion":"0.00000024","prompt":"0.00000006"},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":300000,"is_moderated":true,"max_completion_tokens":5120}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Nova"},"canonical_slug":"amazon/nova-micro-v1","context_length":128000,"created":1733437237,"default_parameters":{},"description":"Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...","expiration_date":null,"hugging_face_id":"","id":"amazon/nova-micro-v1","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/amazon/nova-micro-v1/endpoints"},"name":"Amazon: Nova Micro 1.0","per_request_limits":null,"pricing":{"completion":"0.00000014","prompt":"0.000000035"},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":5120}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Nova"},"canonical_slug":"amazon/nova-pro-v1","context_length":300000,"created":1733436303,"default_parameters":{},"description":"Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...","expiration_date":null,"hugging_face_id":"","id":"amazon/nova-pro-v1","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/amazon/nova-pro-v1/endpoints"},"name":"Amazon: Nova Pro 1.0","per_request_limits":null,"pricing":{"completion":"0.0000032","prompt":"0.0000008"},"supported_parameters":["max_tokens","stop","temperature","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":300000,"is_moderated":true,"max_completion_tokens":5120}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-2024-11-20","context_length":128000,"created":1732127594,"default_parameters":{},"description":"The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...","expiration_date":null,"hugging_face_id":"","id":"openai/gpt-4o-2024-11-20","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-2024-11-20/endpoints"},"name":"OpenAI: GPT-4o (2024-11-20)","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.00000125","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-large-2411","context_length":131072,"created":1731978685,"default_parameters":{"temperature":0.3},"description":"Mistral Large 2 2411 is an update of [Mistral Large 2](/mistralai/mistral-large) released together with [Pixtral Large 2411](/mistralai/pixtral-large-2411) It provides a significant upgrade on the previous [Mistral Large 24.07](/mistralai/mistral-large-2407), with notable...","expiration_date":null,"hugging_face_id":"","id":"mistralai/mistral-large-2411","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/mistralai/mistral-large-2411/endpoints"},"name":"Mistral Large 2411","per_request_limits":null,"pricing":{"completion":"0.000006","input_cache_read":"0.0000002","prompt":"0.000002"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-large-2407","context_length":131072,"created":1731978415,"default_parameters":{"temperature":0.3},"description":"This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","expiration_date":null,"hugging_face_id":"","id":"mistralai/mistral-large-2407","knowledge_cutoff":"2024-03-31","links":{"details":"/api/v1/models/mistralai/mistral-large-2407/endpoints"},"name":"Mistral Large 2407","per_request_limits":null,"pricing":{"completion":"0.000006","input_cache_read":"0.0000002","prompt":"0.000002"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/pixtral-large-2411","context_length":131072,"created":1731977388,"default_parameters":{"temperature":0.3},"description":"Pixtral Large is a 124B parameter, open-weight, multimodal model built on top of [Mistral Large 2](/mistralai/mistral-large-2411). The model is able to understand documents, charts and natural images. The model is...","expiration_date":null,"hugging_face_id":"","id":"mistralai/pixtral-large-2411","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/mistralai/pixtral-large-2411/endpoints"},"name":"Mistral: Pixtral Large 2411","per_request_limits":null,"pricing":{"completion":"0.000006","input_cache_read":"0.0000002","prompt":"0.000002"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-2.5-coder-32b-instruct","context_length":32768,"created":1731368400,"default_parameters":{},"description":"Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...","expiration_date":null,"hugging_face_id":"Qwen/Qwen2.5-Coder-32B-Instruct","id":"qwen/qwen-2.5-coder-32b-instruct","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/qwen/qwen-2.5-coder-32b-instruct/endpoints"},"name":"Qwen2.5 Coder 32B Instruct","per_request_limits":null,"pricing":{"completion":"0.000001","prompt":"0.00000066"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"mistral","modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"thedrummer/unslopnemo-12b","context_length":32768,"created":1731103448,"default_parameters":{},"description":"UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.","expiration_date":null,"hugging_face_id":"TheDrummer/UnslopNemo-12B-v4.1","id":"thedrummer/unslopnemo-12b","knowledge_cutoff":"2024-04-30","links":{"details":"/api/v1/models/thedrummer/unslopnemo-12b/endpoints"},"name":"TheDrummer: UnslopNemo 12B","per_request_limits":null,"pricing":{"completion":"0.0000004","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","logprobs","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-3-5-haiku","context_length":200000,"created":1730678400,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Claude 3.5 Haiku features offers enhanced capabilities in speed, coding accuracy, and tool use. Engineered to excel in real-time applications, it delivers quick response times that are essential for dynamic...","expiration_date":null,"hugging_face_id":null,"id":"anthropic/claude-3.5-haiku","knowledge_cutoff":"2024-07-31","links":{"details":"/api/v1/models/anthropic/claude-3-5-haiku/endpoints"},"name":"Anthropic: Claude 3.5 Haiku","per_request_limits":null,"pricing":{"completion":"0.000004","input_cache_read":"0.00000008","input_cache_write":"0.000001","prompt":"0.0000008","web_search":"0.01"},"supported_parameters":["max_tokens","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"anthracite-org/magnum-v4-72b","context_length":16384,"created":1729555200,"default_parameters":{},"description":"This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).","expiration_date":null,"hugging_face_id":"anthracite-org/magnum-v4-72b","id":"anthracite-org/magnum-v4-72b","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/anthracite-org/magnum-v4-72b/endpoints"},"name":"Magnum v4 72B","per_request_limits":null,"pricing":{"completion":"0.000005","prompt":"0.000003"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":16384,"is_moderated":false,"max_completion_tokens":2048}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-2.5-7b-instruct","context_length":32768,"created":1729036800,"default_parameters":{"frequency_penalty":null,"temperature":null,"top_p":null},"description":"Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","expiration_date":null,"hugging_face_id":"Qwen/Qwen2.5-7B-Instruct","id":"qwen/qwen-2.5-7b-instruct","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/qwen/qwen-2.5-7b-instruct/endpoints"},"name":"Qwen: Qwen2.5 7B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000001","prompt":"0.00000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nvidia/llama-3.1-nemotron-70b-instruct","context_length":131072,"created":1728950400,"default_parameters":{},"description":"NVIDIA's Llama 3.1 Nemotron 70B is a language model designed for generating precise and useful responses. Leveraging [Llama 3.1 70B](/models/meta-llama/llama-3.1-70b-instruct) architecture and Reinforcement Learning from Human Feedback (RLHF), it excels...","expiration_date":"2026-05-07","hugging_face_id":"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF","id":"nvidia/llama-3.1-nemotron-70b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/nvidia/llama-3.1-nemotron-70b-instruct/endpoints"},"name":"NVIDIA: Llama 3.1 Nemotron 70B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000012","prompt":"0.0000012"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"inflection/inflection-3-productivity","context_length":8000,"created":1728604800,"default_parameters":{},"description":"Inflection 3 Productivity is optimized for following instructions. It is better for tasks requiring JSON output or precise adherence to provided guidelines. It has access to recent news. For emotional...","expiration_date":null,"hugging_face_id":null,"id":"inflection/inflection-3-productivity","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/inflection/inflection-3-productivity/endpoints"},"name":"Inflection: Inflection 3 Productivity","per_request_limits":null,"pricing":{"completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":8000,"is_moderated":false,"max_completion_tokens":1024}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Other"},"canonical_slug":"inflection/inflection-3-pi","context_length":8000,"created":1728604800,"default_parameters":{},"description":"Inflection 3 Pi powers Inflection's [Pi](https://pi.ai) chatbot, including backstory, emotional intelligence, productivity, and safety. It has access to recent news, and excels in scenarios like customer support and roleplay. Pi...","expiration_date":null,"hugging_face_id":null,"id":"inflection/inflection-3-pi","knowledge_cutoff":"2024-10-31","links":{"details":"/api/v1/models/inflection/inflection-3-pi/endpoints"},"name":"Inflection: Inflection 3 Pi","per_request_limits":null,"pricing":{"completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["max_tokens","stop","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":8000,"is_moderated":false,"max_completion_tokens":1024}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"thedrummer/rocinante-12b","context_length":32768,"created":1727654400,"default_parameters":{},"description":"Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...","expiration_date":null,"hugging_face_id":"TheDrummer/Rocinante-12B-v1.1","id":"thedrummer/rocinante-12b","knowledge_cutoff":"2024-04-30","links":{"details":"/api/v1/models/thedrummer/rocinante-12b/endpoints"},"name":"TheDrummer: Rocinante 12B","per_request_limits":null,"pricing":{"completion":"0.00000043","prompt":"0.00000017"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":32768}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.2-3b-instruct","context_length":131072,"created":1727222400,"default_parameters":{},"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-3.2-3B-Instruct","id":"meta-llama/llama-3.2-3b-instruct:free","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.2-3b-instruct/endpoints"},"name":"Meta: Llama 3.2 3B Instruct (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.2-3b-instruct","context_length":80000,"created":1727222400,"default_parameters":{},"description":"Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-3.2-3B-Instruct","id":"meta-llama/llama-3.2-3b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.2-3b-instruct/endpoints"},"name":"Meta: Llama 3.2 3B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000034","prompt":"0.000000051"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":80000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.2-1b-instruct","context_length":60000,"created":1727222400,"default_parameters":{},"description":"Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-3.2-1B-Instruct","id":"meta-llama/llama-3.2-1b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.2-1b-instruct/endpoints"},"name":"Meta: Llama 3.2 1B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000002","prompt":"0.000000027"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":60000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image"],"instruct_type":"llama3","modality":"text+image->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.2-11b-vision-instruct","context_length":131072,"created":1727222400,"default_parameters":{},"description":"Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and...","expiration_date":null,"hugging_face_id":"meta-llama/Llama-3.2-11B-Vision-Instruct","id":"meta-llama/llama-3.2-11b-vision-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.2-11b-vision-instruct/endpoints"},"name":"Meta: Llama 3.2 11B Vision Instruct","per_request_limits":null,"pricing":{"completion":"0.000000245","prompt":"0.000000245"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Qwen"},"canonical_slug":"qwen/qwen-2.5-72b-instruct","context_length":32768,"created":1726704000,"default_parameters":{},"description":"Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...","expiration_date":null,"hugging_face_id":"Qwen/Qwen2.5-72B-Instruct","id":"qwen/qwen-2.5-72b-instruct","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/qwen/qwen-2.5-72b-instruct/endpoints"},"name":"Qwen2.5 72B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000004","prompt":"0.00000036"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Cohere"},"canonical_slug":"cohere/command-r-plus-08-2024","context_length":128000,"created":1724976000,"default_parameters":{},"description":"command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...","expiration_date":null,"hugging_face_id":null,"id":"cohere/command-r-plus-08-2024","knowledge_cutoff":"2024-03-31","links":{"details":"/api/v1/models/cohere/command-r-plus-08-2024/endpoints"},"name":"Cohere: Command R+ (08-2024)","per_request_limits":null,"pricing":{"completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":4000}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Cohere"},"canonical_slug":"cohere/command-r-08-2024","context_length":128000,"created":1724976000,"default_parameters":{},"description":"command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...","expiration_date":null,"hugging_face_id":null,"id":"cohere/command-r-08-2024","knowledge_cutoff":"2024-03-31","links":{"details":"/api/v1/models/cohere/command-r-08-2024/endpoints"},"name":"Cohere: Command R (08-2024)","per_request_limits":null,"pricing":{"completion":"0.0000006","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":4000}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"sao10k/l3.1-euryale-70b","context_length":131072,"created":1724803200,"default_parameters":{},"description":"Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).","expiration_date":null,"hugging_face_id":"Sao10K/L3.1-70B-Euryale-v2.2","id":"sao10k/l3.1-euryale-70b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/sao10k/l3.1-euryale-70b/endpoints"},"name":"Sao10K: Llama 3.1 Euryale 70B v2.2","per_request_limits":null,"pricing":{"completion":"0.00000085","prompt":"0.00000085"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nousresearch/hermes-3-llama-3.1-70b","context_length":131072,"created":1723939200,"default_parameters":{},"description":"Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","expiration_date":null,"hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-70B","id":"nousresearch/hermes-3-llama-3.1-70b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-70b/endpoints"},"name":"Nous: Hermes 3 70B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000003","prompt":"0.0000003"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","context_length":131072,"created":1723766400,"default_parameters":{},"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","expiration_date":null,"hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-405B","id":"nousresearch/hermes-3-llama-3.1-405b:free","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-405b/endpoints"},"name":"Nous: Hermes 3 405B Instruct (free)","per_request_limits":null,"pricing":{"completion":"0","prompt":"0"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nousresearch/hermes-3-llama-3.1-405b","context_length":131072,"created":1723766400,"default_parameters":{},"description":"Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...","expiration_date":null,"hugging_face_id":"NousResearch/Hermes-3-Llama-3.1-405B","id":"nousresearch/hermes-3-llama-3.1-405b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/nousresearch/hermes-3-llama-3.1-405b/endpoints"},"name":"Nous: Hermes 3 405B Instruct","per_request_limits":null,"pricing":{"completion":"0.000001","prompt":"0.000001"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"sao10k/l3-lunaris-8b","context_length":8192,"created":1723507200,"default_parameters":{},"description":"Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....","expiration_date":null,"hugging_face_id":"Sao10K/L3-8B-Lunaris-v1","id":"sao10k/l3-lunaris-8b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/sao10k/l3-lunaris-8b/endpoints"},"name":"Sao10K: Llama 3 8B Lunaris","per_request_limits":null,"pricing":{"completion":"0.00000005","prompt":"0.00000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":8192,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-2024-08-06","context_length":128000,"created":1722902400,"default_parameters":{},"description":"The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4o-2024-08-06","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-2024-08-06/endpoints"},"name":"OpenAI: GPT-4o (2024-08-06)","per_request_limits":null,"pricing":{"completion":"0.00001","input_cache_read":"0.00000125","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.1-8b-instruct","context_length":16384,"created":1721692800,"default_parameters":{},"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...","expiration_date":null,"hugging_face_id":"meta-llama/Meta-Llama-3.1-8B-Instruct","id":"meta-llama/llama-3.1-8b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.1-8b-instruct/endpoints"},"name":"Meta: Llama 3.1 8B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000005","prompt":"0.00000002"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":16384,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3.1-70b-instruct","context_length":131072,"created":1721692800,"default_parameters":{},"description":"Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...","expiration_date":null,"hugging_face_id":"meta-llama/Meta-Llama-3.1-70B-Instruct","id":"meta-llama/llama-3.1-70b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3.1-70b-instruct/endpoints"},"name":"Meta: Llama 3.1 70B Instruct","per_request_limits":null,"pricing":{"completion":"0.0000004","prompt":"0.0000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"mistral","modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-nemo","context_length":131072,"created":1721347200,"default_parameters":{"temperature":0.3},"description":"A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...","expiration_date":null,"hugging_face_id":"mistralai/Mistral-Nemo-Instruct-2407","id":"mistralai/mistral-nemo","knowledge_cutoff":"2024-04-30","links":{"details":"/api/v1/models/mistralai/mistral-nemo/endpoints"},"name":"Mistral: Mistral Nemo","per_request_limits":null,"pricing":{"completion":"0.00000003","prompt":"0.00000002"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":131072,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-mini-2024-07-18","context_length":128000,"created":1721260800,"default_parameters":{},"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4o-mini-2024-07-18","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-mini-2024-07-18/endpoints"},"name":"OpenAI: GPT-4o-mini (2024-07-18)","per_request_limits":null,"pricing":{"completion":"0.0000006","input_cache_read":"0.000000075","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-mini","context_length":128000,"created":1721260800,"default_parameters":{},"description":"GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4o-mini","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-mini/endpoints"},"name":"OpenAI: GPT-4o-mini","per_request_limits":null,"pricing":{"completion":"0.0000006","input_cache_read":"0.000000075","prompt":"0.00000015"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"gemma","modality":"text->text","output_modalities":["text"],"tokenizer":"Gemini"},"canonical_slug":"google/gemma-2-27b-it","context_length":8192,"created":1720828800,"default_parameters":{},"description":"Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...","expiration_date":null,"hugging_face_id":"google/gemma-2-27b-it","id":"google/gemma-2-27b-it","knowledge_cutoff":"2024-06-30","links":{"details":"/api/v1/models/google/gemma-2-27b-it/endpoints"},"name":"Google: Gemma 2 27B","per_request_limits":null,"pricing":{"completion":"0.00000065","prompt":"0.00000065"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_p"],"supported_voices":null,"top_provider":{"context_length":8192,"is_moderated":false,"max_completion_tokens":2048}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"sao10k/l3-euryale-70b","context_length":8192,"created":1718668800,"default_parameters":{},"description":"Euryale 70B v2.1 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). - Better prompt adherence. - Better anatomy / spatial awareness. - Adapts much better to unique and custom...","expiration_date":null,"hugging_face_id":"Sao10K/L3-70B-Euryale-v2.1","id":"sao10k/l3-euryale-70b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/sao10k/l3-euryale-70b/endpoints"},"name":"Sao10k: Llama 3 Euryale 70B v2.1","per_request_limits":null,"pricing":{"completion":"0.00000148","prompt":"0.00000148"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":8192,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"nousresearch/hermes-2-pro-llama-3-8b","context_length":8192,"created":1716768000,"default_parameters":{},"description":"Hermes 2 Pro is an upgraded, retrained version of Nous Hermes 2, consisting of an updated and cleaned version of the OpenHermes 2.5 Dataset, as well as a newly introduced...","expiration_date":null,"hugging_face_id":"NousResearch/Hermes-2-Pro-Llama-3-8B","id":"nousresearch/hermes-2-pro-llama-3-8b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/nousresearch/hermes-2-pro-llama-3-8b/endpoints"},"name":"NousResearch: Hermes 2 Pro - Llama-3 8B","per_request_limits":null,"pricing":{"completion":"0.00000014","prompt":"0.00000014"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":8192,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o-2024-05-13","context_length":128000,"created":1715558400,"default_parameters":{},"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4o-2024-05-13","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o-2024-05-13/endpoints"},"name":"OpenAI: GPT-4o (2024-05-13)","per_request_limits":null,"pricing":{"completion":"0.000015","prompt":"0.000005"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text","image","file"],"instruct_type":null,"modality":"text+image+file->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4o","context_length":128000,"created":1715558400,"default_parameters":{},"description":"GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4o","knowledge_cutoff":"2023-10-31","links":{"details":"/api/v1/models/openai/gpt-4o/endpoints"},"name":"OpenAI: GPT-4o","per_request_limits":null,"pricing":{"completion":"0.00001","prompt":"0.0000025"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3-8b-instruct","context_length":8192,"created":1713398400,"default_parameters":{},"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 8B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...","expiration_date":null,"hugging_face_id":"meta-llama/Meta-Llama-3-8B-Instruct","id":"meta-llama/llama-3-8b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3-8b-instruct/endpoints"},"name":"Meta: Llama 3 8B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000004","prompt":"0.00000004"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":8192,"is_moderated":false,"max_completion_tokens":8192}},{"architecture":{"input_modalities":["text"],"instruct_type":"llama3","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama3"},"canonical_slug":"meta-llama/llama-3-70b-instruct","context_length":8192,"created":1713398400,"default_parameters":{},"description":"Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 70B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...","expiration_date":null,"hugging_face_id":"meta-llama/Meta-Llama-3-70B-Instruct","id":"meta-llama/llama-3-70b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/meta-llama/llama-3-70b-instruct/endpoints"},"name":"Meta: Llama 3 70B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000074","prompt":"0.00000051"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":8192,"is_moderated":false,"max_completion_tokens":8000}},{"architecture":{"input_modalities":["text"],"instruct_type":"mistral","modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mixtral-8x22b-instruct","context_length":65536,"created":1713312000,"default_parameters":{"temperature":0.3},"description":"Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...","expiration_date":null,"hugging_face_id":"mistralai/Mixtral-8x22B-Instruct-v0.1","id":"mistralai/mixtral-8x22b-instruct","knowledge_cutoff":"2024-01-31","links":{"details":"/api/v1/models/mistralai/mixtral-8x22b-instruct/endpoints"},"name":"Mistral: Mixtral 8x22B Instruct","per_request_limits":null,"pricing":{"completion":"0.000006","input_cache_read":"0.0000002","prompt":"0.000002"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":65536,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":"vicuna","modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"microsoft/wizardlm-2-8x22b","context_length":65535,"created":1713225600,"default_parameters":{},"description":"WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...","expiration_date":null,"hugging_face_id":"microsoft/WizardLM-2-8x22B","id":"microsoft/wizardlm-2-8x22b","knowledge_cutoff":"2024-04-30","links":{"details":"/api/v1/models/microsoft/wizardlm-2-8x22b/endpoints"},"name":"WizardLM-2 8x22B","per_request_limits":null,"pricing":{"completion":"0.00000062","prompt":"0.00000062"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","stop","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":65535,"is_moderated":false,"max_completion_tokens":8000}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4-turbo","context_length":128000,"created":1712620800,"default_parameters":{},"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4-turbo","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/openai/gpt-4-turbo/endpoints"},"name":"OpenAI: GPT-4 Turbo","per_request_limits":null,"pricing":{"completion":"0.00003","prompt":"0.00001"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text","image"],"instruct_type":null,"modality":"text+image->text","output_modalities":["text"],"tokenizer":"Claude"},"canonical_slug":"anthropic/claude-3-haiku","context_length":200000,"created":1710288000,"default_parameters":{},"description":"Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal","expiration_date":null,"hugging_face_id":null,"id":"anthropic/claude-3-haiku","knowledge_cutoff":"2023-08-31","links":{"details":"/api/v1/models/anthropic/claude-3-haiku/endpoints"},"name":"Anthropic: Claude 3 Haiku","per_request_limits":null,"pricing":{"completion":"0.00000125","input_cache_read":"0.00000003","input_cache_write":"0.0000003","prompt":"0.00000025"},"supported_parameters":["max_tokens","stop","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":200000,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-large","context_length":128000,"created":1708905600,"default_parameters":{"temperature":0.3},"description":"This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....","expiration_date":null,"hugging_face_id":null,"id":"mistralai/mistral-large","knowledge_cutoff":"2024-11-30","links":{"details":"/api/v1/models/mistralai/mistral-large/endpoints"},"name":"Mistral Large","per_request_limits":null,"pricing":{"completion":"0.000006","input_cache_read":"0.0000002","prompt":"0.000002"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4-turbo-preview","context_length":128000,"created":1706140800,"default_parameters":{},"description":"The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4-turbo-preview","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/openai/gpt-4-turbo-preview/endpoints"},"name":"OpenAI: GPT-4 Turbo Preview","per_request_limits":null,"pricing":{"completion":"0.00003","prompt":"0.00001"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-3.5-turbo-0613","context_length":4095,"created":1706140800,"default_parameters":{},"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-3.5-turbo-0613","knowledge_cutoff":"2021-09-30","links":{"details":"/api/v1/models/openai/gpt-3.5-turbo-0613/endpoints"},"name":"OpenAI: GPT-3.5 Turbo (older v0613)","per_request_limits":null,"pricing":{"completion":"0.000002","prompt":"0.000001"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":4095,"is_moderated":false,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":"mistral","modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mixtral-8x7b-instruct","context_length":32768,"created":1702166400,"default_parameters":{"temperature":0.3},"description":"Mixtral 8x7B Instruct is a pretrained generative Sparse Mixture of Experts, by Mistral AI, for chat and instruction use. Incorporates 8 experts (feed-forward networks) for a total of 47 billion...","expiration_date":"2026-05-07","hugging_face_id":"mistralai/Mixtral-8x7B-Instruct-v0.1","id":"mistralai/mixtral-8x7b-instruct","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/mistralai/mixtral-8x7b-instruct/endpoints"},"name":"Mistral: Mixtral 8x7B Instruct","per_request_limits":null,"pricing":{"completion":"0.00000054","prompt":"0.00000054"},"supported_parameters":["frequency_penalty","logit_bias","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":32768,"is_moderated":false,"max_completion_tokens":16384}},{"architecture":{"input_modalities":["text"],"instruct_type":"airoboros","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama2"},"canonical_slug":"alpindale/goliath-120b","context_length":6144,"created":1699574400,"default_parameters":{},"description":"A large LLM created by combining two fine-tuned Llama 70B models into one 120B model. Combines Xwin and Euryale. Credits to - [@chargoddard](https://huggingface.co/chargoddard) for developing the framework used to merge...","expiration_date":null,"hugging_face_id":"alpindale/goliath-120b","id":"alpindale/goliath-120b","knowledge_cutoff":"2023-12-31","links":{"details":"/api/v1/models/alpindale/goliath-120b/endpoints"},"name":"Goliath 120B","per_request_limits":null,"pricing":{"completion":"0.0000075","prompt":"0.00000375"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":6144,"is_moderated":false,"max_completion_tokens":1024}},{"architecture":{"input_modalities":["text","image","audio","file","video"],"instruct_type":null,"modality":"text+image+file+audio+video->text+image","output_modalities":["text","image"],"tokenizer":"Router"},"canonical_slug":"openrouter/auto","context_length":2000000,"created":1699401600,"default_parameters":{"frequency_penalty":null,"presence_penalty":null,"repetition_penalty":null,"temperature":null,"top_k":null,"top_p":null},"description":"\"Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...","expiration_date":null,"hugging_face_id":null,"id":"openrouter/auto","knowledge_cutoff":null,"links":{"details":"/api/v1/models/openrouter/auto/endpoints"},"name":"Auto Router","per_request_limits":null,"pricing":{"completion":"-1","prompt":"-1"},"supported_parameters":["frequency_penalty","include_reasoning","logit_bias","logprobs","max_completion_tokens","max_tokens","min_p","presence_penalty","reasoning","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_k","top_logprobs","top_p","web_search_options"],"supported_voices":null,"top_provider":{"context_length":null,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4-1106-preview","context_length":128000,"created":1699228800,"default_parameters":{},"description":"The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to April 2023.","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4-1106-preview","knowledge_cutoff":"2023-04-30","links":{"details":"/api/v1/models/openai/gpt-4-1106-preview/endpoints"},"name":"OpenAI: GPT-4 Turbo (older v1106)","per_request_limits":null,"pricing":{"completion":"0.00003","prompt":"0.00001"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":128000,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":"chatml","modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-3.5-turbo-instruct","context_length":4095,"created":1695859200,"default_parameters":{},"description":"This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-3.5-turbo-instruct","knowledge_cutoff":"2021-09-30","links":{"details":"/api/v1/models/openai/gpt-3.5-turbo-instruct/endpoints"},"name":"OpenAI: GPT-3.5 Turbo Instruct","per_request_limits":null,"pricing":{"completion":"0.000002","prompt":"0.0000015"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":4095,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":"mistral","modality":"text->text","output_modalities":["text"],"tokenizer":"Mistral"},"canonical_slug":"mistralai/mistral-7b-instruct-v0.1","context_length":2824,"created":1695859200,"default_parameters":{"temperature":0.3},"description":"A 7.3B parameter model that outperforms Llama 2 13B on all benchmarks, with optimizations for speed and context length.","expiration_date":"2026-05-30","hugging_face_id":"mistralai/Mistral-7B-Instruct-v0.1","id":"mistralai/mistral-7b-instruct-v0.1","knowledge_cutoff":"2023-09-30","links":{"details":"/api/v1/models/mistralai/mistral-7b-instruct-v0.1/endpoints"},"name":"Mistral: Mistral 7B Instruct v0.1","per_request_limits":null,"pricing":{"completion":"0.00000019","prompt":"0.00000011"},"supported_parameters":["frequency_penalty","max_tokens","presence_penalty","repetition_penalty","seed","temperature","top_k","top_p"],"supported_voices":null,"top_provider":{"context_length":2824,"is_moderated":false,"max_completion_tokens":null}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-3.5-turbo-16k","context_length":16385,"created":1693180800,"default_parameters":{},"description":"This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-3.5-turbo-16k","knowledge_cutoff":"2021-09-30","links":{"details":"/api/v1/models/openai/gpt-3.5-turbo-16k/endpoints"},"name":"OpenAI: GPT-3.5 Turbo 16k","per_request_limits":null,"pricing":{"completion":"0.000004","prompt":"0.000003"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":16385,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":"alpaca","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama2"},"canonical_slug":"mancer/weaver","context_length":8000,"created":1690934400,"default_parameters":{},"description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","expiration_date":null,"hugging_face_id":null,"id":"mancer/weaver","knowledge_cutoff":"2023-06-30","links":{"details":"/api/v1/models/mancer/weaver/endpoints"},"name":"Mancer: Weaver (alpha)","per_request_limits":null,"pricing":{"completion":"0.000001","prompt":"0.00000075"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","temperature","top_a","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":8000,"is_moderated":false,"max_completion_tokens":2000}},{"architecture":{"input_modalities":["text"],"instruct_type":"alpaca","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama2"},"canonical_slug":"undi95/remm-slerp-l2-13b","context_length":6144,"created":1689984000,"default_parameters":{},"description":"A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge","expiration_date":null,"hugging_face_id":"Undi95/ReMM-SLERP-L2-13B","id":"undi95/remm-slerp-l2-13b","knowledge_cutoff":"2023-06-30","links":{"details":"/api/v1/models/undi95/remm-slerp-l2-13b/endpoints"},"name":"ReMM SLERP 13B","per_request_limits":null,"pricing":{"completion":"0.00000065","prompt":"0.00000045"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":6144,"is_moderated":false,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":"alpaca","modality":"text->text","output_modalities":["text"],"tokenizer":"Llama2"},"canonical_slug":"gryphe/mythomax-l2-13b","context_length":4096,"created":1688256000,"default_parameters":{},"description":"One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge","expiration_date":null,"hugging_face_id":"Gryphe/MythoMax-L2-13b","id":"gryphe/mythomax-l2-13b","knowledge_cutoff":"2023-06-30","links":{"details":"/api/v1/models/gryphe/mythomax-l2-13b/endpoints"},"name":"MythoMax 13B","per_request_limits":null,"pricing":{"completion":"0.00000006","prompt":"0.00000006"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","min_p","presence_penalty","repetition_penalty","response_format","seed","stop","structured_outputs","temperature","top_a","top_k","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":4096,"is_moderated":false,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4-0314","context_length":8191,"created":1685232000,"default_parameters":{},"description":"GPT-4-0314 is the first version of GPT-4 released, with a context length of 8,192 tokens, and was supported until June 14. Training data: up to Sep 2021.","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4-0314","knowledge_cutoff":"2021-09-30","links":{"details":"/api/v1/models/openai/gpt-4-0314/endpoints"},"name":"OpenAI: GPT-4 (older v0314)","per_request_limits":null,"pricing":{"completion":"0.00006","prompt":"0.00003"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":8191,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-4","context_length":8191,"created":1685232000,"default_parameters":{},"description":"OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-4","knowledge_cutoff":"2021-09-30","links":{"details":"/api/v1/models/openai/gpt-4/endpoints"},"name":"OpenAI: GPT-4","per_request_limits":null,"pricing":{"completion":"0.00006","prompt":"0.00003"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_completion_tokens","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":8191,"is_moderated":true,"max_completion_tokens":4096}},{"architecture":{"input_modalities":["text"],"instruct_type":null,"modality":"text->text","output_modalities":["text"],"tokenizer":"GPT"},"canonical_slug":"openai/gpt-3.5-turbo","context_length":16385,"created":1685232000,"default_parameters":{},"description":"GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.","expiration_date":null,"hugging_face_id":null,"id":"openai/gpt-3.5-turbo","knowledge_cutoff":"2021-09-30","links":{"details":"/api/v1/models/openai/gpt-3.5-turbo/endpoints"},"name":"OpenAI: GPT-3.5 Turbo","per_request_limits":null,"pricing":{"completion":"0.0000015","prompt":"0.0000005"},"supported_parameters":["frequency_penalty","logit_bias","logprobs","max_tokens","presence_penalty","response_format","seed","stop","structured_outputs","temperature","tool_choice","tools","top_logprobs","top_p"],"supported_voices":null,"top_provider":{"context_length":16385,"is_moderated":true,"max_completion_tokens":4096}}]}
