{
  "302ai": {
    "api": "https://api.302.ai/v1",
    "doc": "https://doc.302.ai",
    "env": [
      "302AI_API_KEY"
    ],
    "id": "302ai",
    "models": {
      "MiniMax-M1": {
        "attachment": false,
        "cost": {
          "input": 0.132,
          "output": 1.254
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M1",
        "last_updated": "2025-06-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-16",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0.33,
          "output": 1.32
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "MiniMax-M2",
        "last_updated": "2025-10-26",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-26",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "MiniMax-M2.1",
        "last_updated": "2025-12-19",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-19",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "MiniMax-M2.7",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-19",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7-highspeed": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 4.8
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "id": "MiniMax-M2.7-highspeed",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-19",
        "temperature": true,
        "tool_call": true
      },
      "chatgpt-4o-latest": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 15
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "chatgpt-4o-latest",
        "knowledge": "2023-09",
        "last_updated": "2024-08-08",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "chatgpt-4o-latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-08",
        "temperature": true,
        "tool_call": false
      },
      "claude-3-5-haiku-20241022": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-3-5-haiku-20241022",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-3-5-haiku-20241022",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-5-haiku-latest": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-3-5-haiku-latest",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-3-5-haiku-latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-16",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-haiku-4-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-16",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-16",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-haiku-4-5-20251001",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-16",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1-20250805",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-1-20250805",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805-thinking": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "claude-opus-4-1-20250805-thinking",
        "knowledge": "2025-03",
        "last_updated": "2025-05-27",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-1-20250805-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-27",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-20250514",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5-20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-5-20251101",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101-thinking": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "claude-opus-4-5-20251101-thinking",
        "knowledge": "2025-03",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-5-20251101-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-06",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6-thinking": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "claude-opus-4-6-thinking",
        "knowledge": "2025-05",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-6-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-06",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-20250514",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-5-20250929",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929-thinking": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "claude-sonnet-4-5-20250929-thinking",
        "knowledge": "2025-03",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-5-20250929-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-18",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6-thinking": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "claude-sonnet-4-6-thinking",
        "knowledge": "2025-08",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-6-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-18",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-chat": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.43
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-chat",
        "knowledge": "2024-07",
        "last_updated": "2024-11-29",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek-Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-29",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-reasoner": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.43
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-reasoner",
        "knowledge": "2024-07",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek-Reasoner",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.43
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek-v3.2",
        "knowledge": "2024-12",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-v3.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.43
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek-v3.2-thinking",
        "knowledge": "2024-12",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-1-6-thinking-250715": {
        "attachment": true,
        "cost": {
          "input": 0.121,
          "output": 1.21
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-1-6-thinking-250715",
        "last_updated": "2025-07-15",
        "limit": {
          "context": 256000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "doubao-seed-1-6-thinking-250715",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-15",
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-1-6-vision-250815": {
        "attachment": true,
        "cost": {
          "input": 0.114,
          "output": 1.143
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "doubao-seed-1-6-vision-250815",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "doubao-seed-1-6-vision-250815",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-1-8-251215": {
        "attachment": true,
        "cost": {
          "input": 0.114,
          "output": 0.286
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "doubao-seed-1-8-251215",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 224000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "doubao-seed-1-8-251215",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-18",
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-code-preview-251028": {
        "attachment": true,
        "cost": {
          "input": 0.17,
          "output": 1.14
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "doubao-seed-code-preview-251028",
        "last_updated": "2025-11-11",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "doubao-seed-code-preview-251028",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-11",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.0-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-2.0-flash-lite",
        "knowledge": "2024-11",
        "last_updated": "2025-06-16",
        "limit": {
          "context": 2000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.0-flash-lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-16",
        "temperature": true,
        "tool_call": false
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-image": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 30
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-2.5-flash-image",
        "knowledge": "2025-01",
        "last_updated": "2025-10-08",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-08",
        "temperature": true,
        "tool_call": false
      },
      "gemini-2.5-flash-lite-preview-09-2025": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "gemini-2.5-flash-lite-preview-09-2025",
        "knowledge": "2025-01",
        "last_updated": "2025-09-26",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-lite-preview-09-2025",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-26",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-nothink": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash-nothink",
        "knowledge": "2025-01",
        "last_updated": "2025-06-24",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-nothink",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-24",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-preview-09-2025": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "gemini-2.5-flash-preview-09-2025",
        "knowledge": "2025-01",
        "last_updated": "2025-09-26",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-preview-09-2025",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-26",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-06",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-3-flash-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-18",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro-image-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 120
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-3-pro-image-preview",
        "knowledge": "2025-06",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 32768,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-3-pro-image-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": false
      },
      "gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "gemini-3-pro-preview",
        "knowledge": "2025-06",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-3-pro-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-19",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-image-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 60
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-3.1-flash-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-27",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "gemini-3.1-flash-image-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-27",
        "temperature": true,
        "tool_call": false
      },
      "glm-4.5": {
        "attachment": false,
        "cost": {
          "input": 0.286,
          "output": 1.142
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "input": 0.1143,
          "output": 0.286
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.5-air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-airx": {
        "attachment": false,
        "cost": {
          "input": 0.572,
          "output": 1.714
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.5-airx",
        "knowledge": "2025-04",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.5-airx",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-x": {
        "attachment": false,
        "cost": {
          "input": 1.143,
          "output": 2.29
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.5-x",
        "knowledge": "2025-04",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.5-x",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5v": {
        "attachment": true,
        "cost": {
          "input": 0.29,
          "output": 0.86
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.5v",
        "knowledge": "2025-04",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 64000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "input": 0.286,
          "output": 1.142
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6v": {
        "attachment": true,
        "cost": {
          "input": 0.145,
          "output": 0.43
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.286,
          "output": 1.142
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "input": 0.0715,
          "output": 0.429
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "glm-4.7-flashx",
        "knowledge": "2025-04",
        "last_updated": "2026-01-20",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.7-flashx",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-20",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.6
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 3.2
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-5-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-16",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-5-turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "input": 0.86,
          "output": 3.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-10",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "input": 0.72,
          "output": 3.2
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-02",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "glm-for-coding": {
        "attachment": false,
        "cost": {
          "input": 0.086,
          "output": 0.343
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-for-coding",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-for-coding",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-4.1-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-4.1-nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-08",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-08",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-08",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-08",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-thinking": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "gpt-5-thinking",
        "knowledge": "2024-10",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-08",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-chat-latest": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.1-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.1-chat-latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-12",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-chat-latest": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.2-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-12",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.2-chat-latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-12-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 0,
          "context_over_200k": {
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini-2026-03-17": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini-2026-03-17",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4-mini-2026-03-17",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4-nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano-2026-03-17": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano-2026-03-17",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4-nano-2026-03-17",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "grok-4-1-fast-non-reasoning",
        "knowledge": "2025-06",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-1-fast-non-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "grok-4-1-fast-reasoning",
        "knowledge": "2025-06",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-1-fast-reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "grok-4-fast-non-reasoning",
        "knowledge": "2025-06",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-fast-non-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "grok-4-fast-reasoning",
        "knowledge": "2025-06",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-fast-reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "grok-4.1": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 10
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "grok-4.1",
        "knowledge": "2025-06",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-18",
        "temperature": true,
        "tool_call": true
      },
      "grok-4.20-beta-0309-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "grok-4.20-beta-0309-non-reasoning",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4.20-beta-0309-non-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "grok-4.20-beta-0309-reasoning": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "grok-4.20-beta-0309-reasoning",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4.20-beta-0309-reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "grok-4.20-multi-agent-beta-0309": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "grok-4.20-multi-agent-beta-0309",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4.20-multi-agent-beta-0309",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-0905-preview": {
        "attachment": false,
        "cost": {
          "input": 0.632,
          "output": 2.53
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "kimi-k2-0905-preview",
        "knowledge": "2025-06",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2-0905-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.575,
          "output": 2.3
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "kimi-k2-thinking",
        "knowledge": "2025-06",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking-turbo": {
        "attachment": false,
        "cost": {
          "input": 1.265,
          "output": 9.119
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "kimi-k2-thinking-turbo",
        "knowledge": "2025-06",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2-thinking-turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "ministral-14b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.33,
          "output": 0.33
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "id": "ministral-14b-2512",
        "knowledge": "2024-12",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ministral-14b-2512",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-2512": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 3.3
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "id": "mistral-large-2512",
        "knowledge": "2024-12",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 128000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "mistral-large-2512",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen-flash": {
        "attachment": false,
        "cost": {
          "input": 0.022,
          "output": 0.22
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "id": "qwen-flash",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen-max-latest": {
        "attachment": false,
        "cost": {
          "input": 0.343,
          "output": 1.372
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen-max-latest",
        "knowledge": "2024-11",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-Max-Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-03",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 1.2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-plus",
        "knowledge": "2024-10",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 2.86
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b",
        "knowledge": "2025-04",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 1.143
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "qwen3-235b-a22b-instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-235b-a22b-instruct-2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-30",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.11,
          "output": 1.08
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-30b-a3b",
        "knowledge": "2025-04",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-30B-A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.86,
          "output": 3.43
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-coder-480b-a35b-instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max-2025-09-23": {
        "attachment": false,
        "cost": {
          "input": 0.86,
          "output": 3.43
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "qwen3-max-2025-09-23",
        "knowledge": "2025-04",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 258048,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-max-2025-09-23",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-24",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "302.AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "abacus": {
    "api": "https://routellm.abacus.ai/v1",
    "doc": "https://abacus.ai/help/api",
    "env": [
      "ABACUS_API_KEY"
    ],
    "id": "abacus",
    "models": {
      "Qwen/QwQ-32B": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/QwQ-32B",
        "last_updated": "2024-11-28",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwQ 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-11-28",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-72B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.11,
          "output": 0.38
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-72B-Instruct",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-19",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-32B": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.29
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-32B",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 1.2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/qwen3-coder-480b-a35b-instruct",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-7-sonnet-20250219": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-3-7-sonnet-20250219",
        "knowledge": "2024-10-31",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1-20250805",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-20250514",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-14",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5-20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-20250514",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-14",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 7
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1-Terminus": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1-Terminus",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Terminus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.4
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "last_updated": "2025-06-15",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-15",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 1.66
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-25",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 1,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash",
        "id": "gemini-3.1-flash-lite-preview",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 Nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o-2024-11-20",
        "knowledge": "2024-10",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4o-mini",
        "knowledge": "2024-04",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-codex": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-chat-latest": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5.1-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-chat-latest": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5.2-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2026-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-01",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-chat-latest": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5.3-chat-latest",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-01",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex-xhigh": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5.3-codex-xhigh",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex XHigh",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "grok-4-0709": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-0709",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-09",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-non-reasoning",
        "last_updated": "2025-11-17",
        "limit": {
          "context": 2000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-17",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-non-reasoning",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 2000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4 Fast (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "temperature": true,
        "tool_call": true
      },
      "grok-code-fast-1": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-code-fast-1",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Code Fast 1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-turbo-preview": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 8
        },
        "description": "Fast Kimi model for responsive chat, coding help, and agent loops",
        "family": "kimi-k2",
        "id": "kimi-k2-turbo-preview",
        "last_updated": "2025-07-08",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Turbo Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-08",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-versatile": {
        "attachment": false,
        "cost": {
          "input": 0.59,
          "output": 0.79
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-versatile",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Versatile",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.59
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo": {
        "attachment": false,
        "cost": {
          "input": 3.5,
          "output": 3.5
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 405B Instruct Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Meta-Llama-3.1-8B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Meta-Llama-3.1-8B-Instruct",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "o3": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-20",
        "temperature": false,
        "tool_call": true
      },
      "o3-pro": {
        "attachment": true,
        "cost": {
          "input": 20,
          "output": 40
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "o3-pro",
        "knowledge": "2024-05",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-10",
        "temperature": false,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "output": 0.44
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen-2.5-coder-32b": {
        "attachment": false,
        "cost": {
          "input": 0.79,
          "output": 0.79
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen-2.5-coder-32b",
        "last_updated": "2024-11-11",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 Coder 32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 1.2,
          "output": 6
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3-max",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "route-llm": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "gpt",
        "id": "route-llm",
        "knowledge": "2024-10",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Route LLM",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.5",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.6": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.6",
        "last_updated": "2025-03-01",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-01",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.7",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-5": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Abacus",
    "npm": "@ai-sdk/openai-compatible"
  },
  "abliteration-ai": {
    "api": "https://api.abliteration.ai/v1",
    "doc": "https://docs.abliteration.ai/models",
    "env": [
      "ABLIT_KEY"
    ],
    "id": "abliteration-ai",
    "models": {
      "abliterated-model": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 3
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "abliterated-model",
        "last_updated": "2026-01-06",
        "limit": {
          "context": 150000,
          "input": 150000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Abliterated Model",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "abliteration.ai",
    "npm": "@ai-sdk/openai-compatible"
  },
  "aihubmix": {
    "doc": "https://docs.aihubmix.com",
    "env": [
      "AIHUBMIX_API_KEY"
    ],
    "id": "aihubmix",
    "models": {
      "alicloud-deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek-flash",
        "id": "alicloud-deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash (Alibaba Cloud)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "alicloud-deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 1.69,
          "output": 3.38
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "alicloud-deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro (Alibaba Cloud)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "alicloud-glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.169,
          "cache_write": 1.05625,
          "input": 0.84,
          "output": 3.38
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "alicloud-glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1 (Alibaba Cloud)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "interleaved": true,
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6-think": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6-think",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "interleaved": true,
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-7-think": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-7-think",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "interleaved": true,
        "last_updated": "2026-05-28",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8-think": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8-think",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-28",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "interleaved": true,
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6-think": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6-think",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "coding-glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.013,
          "input": 0.06,
          "output": 0.22
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "coding-glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-11",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "coding-glm-5.1-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm-free",
        "id": "coding-glm-5.1-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-11",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding GLM 5.1 (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "coding-minimax-m2.7": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "coding-minimax-m2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 128100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding MiniMax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "coding-minimax-m2.7-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax-free",
        "id": "coding-minimax-m2.7-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 128100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding MiniMax M2.7 (Free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "coding-minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "coding-minimax-m2.7-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 128100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding MiniMax M2.7 Highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "coding-xiaomi-mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.016,
          "context_over_200k": {
            "cache_read": 0.032,
            "input": 0.16,
            "output": 0.8
          },
          "input": 0.08,
          "output": 0.4,
          "tiers": [
            {
              "cache_read": 0.032,
              "input": 0.16,
              "output": 0.8,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo-v2.5",
        "id": "coding-xiaomi-mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Xiaomi MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "coding-xiaomi-mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "context_over_200k": {
            "cache_read": 0.08,
            "input": 0.4,
            "output": 1.2
          },
          "input": 0.2,
          "output": 0.6,
          "tiers": [
            {
              "cache_read": 0.08,
              "input": 0.4,
              "output": 1.2,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo-v2.5-pro",
        "id": "coding-xiaomi-mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Xiaomi MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "deep-deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0308,
          "input": 0.154,
          "output": 0.308
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek-flash",
        "id": "deep-deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash (DeepSeek)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deep-deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.004302,
          "input": 0.478,
          "output": 0.956
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "deep-deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro (DeepSeek)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2-0-code-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09644,
          "input": 0.48,
          "output": 2.41,
          "tiers": [
            {
              "cache_read": 0.144656,
              "input": 0.72,
              "output": 3.62,
              "tier": {
                "size": 32000,
                "type": "context"
              }
            },
            {
              "cache_read": 0.28932,
              "input": 1.45,
              "output": 7.23,
              "tier": {
                "size": 128000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "seed",
        "id": "doubao-seed-2-0-code-preview",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Code Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2-0-lite-260428": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01692,
          "input": 0.08,
          "input_audio": 1.269,
          "output": 0.51,
          "tiers": [
            {
              "cache_read": 0.02536,
              "input": 0.13,
              "input_audio": 1.902,
              "output": 0.76,
              "tier": {
                "size": 32000,
                "type": "context"
              }
            },
            {
              "cache_read": 0.05072,
              "input": 0.25,
              "input_audio": 3.804,
              "output": 1.52,
              "tier": {
                "size": 128000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "doubao-seed-2-0-lite-260428",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-28",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Lite 260428",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2-0-mini-260428": {
        "attachment": true,
        "cost": {
          "cache_read": 0.00564,
          "input": 0.03,
          "input_audio": 0.423,
          "output": 0.28,
          "tiers": [
            {
              "cache_read": 0.01128,
              "input": 0.06,
              "input_audio": 0.846,
              "output": 0.56,
              "tier": {
                "size": 32000,
                "type": "context"
              }
            },
            {
              "cache_read": 0.02256,
              "input": 0.11,
              "input_audio": 1.692,
              "output": 1.13,
              "tier": {
                "size": 128000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "doubao-seed-2-0-mini-260428",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-28",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Mini 260428",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2-0-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09644,
          "input": 0.48,
          "output": 2.41,
          "tiers": [
            {
              "cache_read": 0.144656,
              "input": 0.72,
              "output": 3.62,
              "tier": {
                "size": 32000,
                "type": "context"
              }
            },
            {
              "cache_read": 0.28932,
              "input": 1.45,
              "output": 7.23,
              "tier": {
                "size": 128000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "doubao-seed-2-0-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-03-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-03-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "context_over_200k": {
            "cache_read": 0.05,
            "input": 0.5,
            "output": 3
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.05,
              "input": 0.5,
              "output": 3,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 1,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview-customtools",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2817,
          "input": 1.1268,
          "output": 3.9438
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.169008,
          "input": 0.7042,
          "output": 3.09848
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glmv",
        "id": "glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-09",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Vision Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.15,
                "input": 1.5,
                "output": 9
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4.3",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0169,
          "cache_write": 0.21125,
          "context_over_200k": {
            "cache_read": 0.0676,
            "cache_write": 0.845,
            "input": 0.68,
            "output": 4.06
          },
          "input": 0.17,
          "output": 1.01,
          "tiers": [
            {
              "cache_read": 0.0676,
              "cache_write": 0.845,
              "input": 0.68,
              "output": 4.06,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 991000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1268,
          "cache_write": 1.585,
          "input": 1.27,
          "output": 7.61,
          "tiers": [
            {
              "cache_read": 0.2112,
              "cache_write": 2.64,
              "input": 2.11,
              "output": 12.67,
              "tier": {
                "size": 128000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "qwen3.6",
        "id": "qwen3.6-max-preview",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-05-09",
        "limit": {
          "context": 240000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0282,
          "cache_write": 0.3525,
          "context_over_200k": {
            "cache_read": 0.1128,
            "cache_write": 1.41,
            "input": 1.13,
            "output": 6.77
          },
          "input": 0.28,
          "output": 1.69,
          "tiers": [
            {
              "cache_read": 0.1128,
              "cache_write": 1.41,
              "input": 1.13,
              "output": 6.77,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qwen3.6",
        "id": "qwen3.6-plus",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-05-09",
        "limit": {
          "context": 991000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.169,
          "cache_write": 2.1125,
          "input": 1.69,
          "output": 5.07
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-21",
        "limit": {
          "context": 991000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0564,
          "cache_write": 0.3525,
          "input": 0.282,
          "output": 1.128
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 991000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xiaomi-mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.088,
          "context_over_200k": {
            "cache_read": 0.176,
            "input": 0.88,
            "output": 4.4
          },
          "input": 0.44,
          "output": 2.2,
          "tiers": [
            {
              "cache_read": 0.176,
              "input": 0.88,
              "output": 4.4,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo-v2.5",
        "id": "xiaomi-mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi-mimo-v2.5-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo-v2.5",
        "id": "xiaomi-mimo-v2.5-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi MiMo-V2.5 (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi-mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.22,
          "context_over_200k": {
            "cache_read": 0.44,
            "input": 2.2,
            "output": 6.6
          },
          "input": 1.1,
          "output": 3.3,
          "tiers": [
            {
              "cache_read": 0.44,
              "input": 2.2,
              "output": 6.6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo-v2.5-pro",
        "id": "xiaomi-mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi-mimo-v2.5-pro-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo-v2.5-pro",
        "id": "xiaomi-mimo-v2.5-pro-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi MiMo-V2.5-Pro (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "zai-glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.183112,
          "input": 0.845,
          "output": 3.38
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1 (Z.ai)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "AIHubMix",
    "npm": "@aihubmix/ai-sdk-provider"
  },
  "alibaba": {
    "api": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
    "doc": "https://www.alibabacloud.com/help/en/model-studio/models",
    "env": [
      "DASHSCOPE_API_KEY"
    ],
    "id": "alibaba",
    "models": {
      "qvq-max": {
        "attachment": false,
        "cost": {
          "input": 1.2,
          "output": 4.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qvq",
        "id": "qvq-max",
        "knowledge": "2024-04",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QVQ Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen-flash": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.4
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen-max": {
        "attachment": false,
        "cost": {
          "input": 1.6,
          "output": 6.4
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen-max",
        "knowledge": "2024-04",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-03",
        "temperature": true,
        "tool_call": true
      },
      "qwen-mt-plus": {
        "attachment": false,
        "cost": {
          "input": 2.46,
          "output": 7.37
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "family": "qwen",
        "id": "qwen-mt-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-01",
        "limit": {
          "context": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-MT Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01",
        "temperature": true,
        "tool_call": false
      },
      "qwen-mt-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.16,
          "output": 0.49
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "family": "qwen",
        "id": "qwen-mt-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-01",
        "limit": {
          "context": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-MT Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01",
        "temperature": true,
        "tool_call": false
      },
      "qwen-omni-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "input_audio": 4.44,
          "output": 0.27,
          "output_audio": 8.89
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen-omni-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-03-26",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen-Omni Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-19",
        "temperature": true,
        "tool_call": true
      },
      "qwen-omni-turbo-realtime": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "input_audio": 4.44,
          "output": 1.07,
          "output_audio": 8.89
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen-omni-turbo-realtime",
        "knowledge": "2024-04",
        "last_updated": "2025-05-08",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen-Omni Turbo Realtime",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-08",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.2,
          "reasoning": 4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus-character-ja": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-plus-character-ja",
        "knowledge": "2024-04",
        "last_updated": "2024-01",
        "limit": {
          "context": 8192,
          "output": 512
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus Character (Japanese)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2,
          "reasoning": 0.5
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-max": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-vl-max",
        "knowledge": "2024-04",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-08",
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-ocr": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "family": "qwen",
        "id": "qwen-vl-ocr",
        "knowledge": "2024-04",
        "last_updated": "2025-04-13",
        "limit": {
          "context": 34096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL OCR",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-28",
        "temperature": true,
        "tool_call": false
      },
      "qwen-vl-plus": {
        "attachment": false,
        "cost": {
          "input": 0.21,
          "output": 0.63
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-vl-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-08-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-14b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 1.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-14b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 14B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.8
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-32b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.4,
          "output": 5.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.175,
          "output": 0.7
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-7b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-omni-7b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "input_audio": 6.76,
          "output": 0.4
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen2-5-omni-7b",
        "knowledge": "2024-04",
        "last_updated": "2024-12",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen2.5-Omni 7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-vl-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2.8,
          "output": 8.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen2-5-vl-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-vl-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 1.05
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen2-5-vl-7b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-14b": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 1.4,
          "reasoning": 4.2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-14b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 14B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.8,
          "reasoning": 8.4
        },
        "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.8,
          "reasoning": 8.4
        },
        "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
        "family": "qwen",
        "id": "qwen3-32b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.18,
          "output": 0.7,
          "reasoning": 2.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-8b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 8B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-asr-flash": {
        "attachment": false,
        "cost": {
          "input": 0.035,
          "output": 0.035
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "qwen",
        "id": "qwen3-asr-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-09-08",
        "limit": {
          "context": 53248,
          "output": 4096
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-ASR Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-08",
        "temperature": false,
        "tool_call": false
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 2.25
        },
        "description": "Smaller Qwen coder for efficient local agents and repo-level fixes",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Open Qwen coding heavyweight for repository reasoning and agentic engineering",
        "family": "qwen",
        "id": "qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 480B-A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-flash",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Hosted Qwen coder for software agents, repo edits, and long-context code",
        "family": "qwen",
        "id": "qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-livetranslate-flash-realtime": {
        "attachment": false,
        "cost": {
          "input": 10,
          "input_audio": 10,
          "output": 10,
          "output_audio": 38
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "qwen",
        "id": "qwen3-livetranslate-flash-realtime",
        "knowledge": "2024-04",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 53248,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen3-LiveTranslate Flash Realtime",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-22",
        "temperature": true,
        "tool_call": false
      },
      "qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 1.2,
          "output": 6
        },
        "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 6
        },
        "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B (Thinking)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-omni-flash": {
        "attachment": false,
        "cost": {
          "input": 0.43,
          "input_audio": 3.81,
          "output": 1.66,
          "output_audio": 15.11
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen3-omni-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen3-Omni Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-omni-flash-realtime": {
        "attachment": false,
        "cost": {
          "input": 0.52,
          "input_audio": 4.57,
          "output": 1.99,
          "output_audio": 18.13
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen3-omni-flash-realtime",
        "knowledge": "2024-04",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen3-Omni Flash Realtime",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-235b-a22b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.8,
          "reasoning": 8.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-235b-a22b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL 235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8,
          "reasoning": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-30b-a3b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL 30B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-plus": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.6,
          "reasoning": 4.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-122b-a10b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-27b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-35b-a3b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen3.5-397b-a17b",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-plus": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2.4,
          "reasoning": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.6-27b",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.248,
          "output": 1.485
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "qwen3.6-35b-a3b",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_write": 0.234375,
          "input": 0.1875,
          "output": 1.125
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "cache_write": 1.625,
          "input": 1.3,
          "output": 7.8
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3.6-max-preview",
        "knowledge": "2025-04",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "qwq-plus": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 2.4
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwq-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-03-05",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwQ Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-05",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Alibaba",
    "npm": "@ai-sdk/openai-compatible"
  },
  "alibaba-cn": {
    "api": "https://dashscope.aliyuncs.com/compatible-mode/v1",
    "doc": "https://www.alibabacloud.com/help/en/model-studio/models",
    "env": [
      "DASHSCOPE_API_KEY"
    ],
    "id": "alibaba-cn",
    "models": {
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax/MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax/MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 2.294
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 2.294
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-0528",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.861
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-distill-llama-70b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Llama 70B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-llama-8b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-distill-llama-8b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Llama 8B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-qwen-1-5b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "deepseek-r1-distill-qwen-1-5b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 1.5B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-qwen-14b": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.431
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "deepseek-r1-distill-qwen-14b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 14B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-qwen-32b": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.861
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "deepseek-r1-distill-qwen-32b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-qwen-7b": {
        "attachment": false,
        "cost": {
          "input": 0.072,
          "output": 0.144
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "deepseek-r1-distill-qwen-7b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 7B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 1.147
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3-1": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 1.721
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3-1",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3-2-exp": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.431
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3-2-exp",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Exp",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "input": 0.86,
          "output": 3.15
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.17,
          "input": 0.87,
          "output": 3.48
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-14",
        "limit": {
          "context": 202752,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.275,
          "cache_write": 0,
          "input": 1.1,
          "output": 3.851
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 2.294
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Moonshot Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 2.411
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Moonshot Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "input": 0.929,
          "output": 3.858
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Moonshot Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "kimi/kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi/kimi-k2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "moonshot-kimi-k2-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 2.294
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshot-kimi-k2-instruct",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Moonshot Kimi K2 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qvq-max": {
        "attachment": false,
        "cost": {
          "input": 1.147,
          "output": 4.588
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qvq",
        "id": "qvq-max",
        "knowledge": "2024-04",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QVQ Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen-deep-research": {
        "attachment": false,
        "cost": {
          "input": 7.742,
          "output": 23.367
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-deep-research",
        "knowledge": "2024-04",
        "last_updated": "2024-01",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Deep Research",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-doc-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.087,
          "output": 0.144
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-doc-turbo",
        "knowledge": "2024-04",
        "last_updated": "2024-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Doc Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-flash": {
        "attachment": false,
        "cost": {
          "input": 0.022,
          "output": 0.216
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen-long": {
        "attachment": false,
        "cost": {
          "input": 0.072,
          "output": 0.287
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-long",
        "knowledge": "2024-04",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 10000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Long",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen-math-plus": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 1.721
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-math-plus",
        "knowledge": "2024-04",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 4096,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Math Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen-math-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.861
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-math-turbo",
        "knowledge": "2024-04",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 4096,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Math Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-19",
        "temperature": true,
        "tool_call": true
      },
      "qwen-max": {
        "attachment": false,
        "cost": {
          "input": 0.345,
          "output": 1.377
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen-max",
        "knowledge": "2024-04",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-03",
        "temperature": true,
        "tool_call": true
      },
      "qwen-mt-plus": {
        "attachment": false,
        "cost": {
          "input": 0.259,
          "output": 0.775
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "family": "qwen",
        "id": "qwen-mt-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-01",
        "limit": {
          "context": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-MT Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01",
        "temperature": true,
        "tool_call": false
      },
      "qwen-mt-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.101,
          "output": 0.28
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "family": "qwen",
        "id": "qwen-mt-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-01",
        "limit": {
          "context": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-MT Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01",
        "temperature": true,
        "tool_call": false
      },
      "qwen-omni-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.058,
          "input_audio": 3.584,
          "output": 0.23,
          "output_audio": 7.168
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen-omni-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-03-26",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen-Omni Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-19",
        "temperature": true,
        "tool_call": true
      },
      "qwen-omni-turbo-realtime": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "input_audio": 3.584,
          "output": 0.918,
          "output_audio": 7.168
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen-omni-turbo-realtime",
        "knowledge": "2024-04",
        "last_updated": "2025-05-08",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen-Omni Turbo Realtime",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-08",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus": {
        "attachment": false,
        "cost": {
          "input": 0.115,
          "output": 0.287,
          "reasoning": 1.147
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus-character": {
        "attachment": false,
        "cost": {
          "input": 0.115,
          "output": 0.287
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-plus-character",
        "knowledge": "2024-04",
        "last_updated": "2024-01",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus Character",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.044,
          "output": 0.087,
          "reasoning": 0.431
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-07-15",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-max": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "output": 0.574
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-vl-max",
        "knowledge": "2024-04",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-08",
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-ocr": {
        "attachment": false,
        "cost": {
          "input": 0.717,
          "output": 0.717
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "family": "qwen",
        "id": "qwen-vl-ocr",
        "knowledge": "2024-04",
        "last_updated": "2025-04-13",
        "limit": {
          "context": 34096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL OCR",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-28",
        "temperature": true,
        "tool_call": false
      },
      "qwen-vl-plus": {
        "attachment": false,
        "cost": {
          "input": 0.115,
          "output": 0.287
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-vl-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-08-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-14b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.431
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-14b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 14B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.861
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-32b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 1.721
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.072,
          "output": 0.144
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-7b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-coder-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.861
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen2-5-coder-32b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-11",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-Coder 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-coder-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.287
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen2-5-coder-7b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-11",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-Coder 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-math-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.574,
          "output": 1.721
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-math-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 4096,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-Math 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-math-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.287
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen2-5-math-7b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 4096,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-Math 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-omni-7b": {
        "attachment": false,
        "cost": {
          "input": 0.087,
          "input_audio": 5.448,
          "output": 0.345
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen2-5-omni-7b",
        "knowledge": "2024-04",
        "last_updated": "2024-12",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen2.5-Omni 7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-vl-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2.294,
          "output": 6.881
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen2-5-vl-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-vl-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.717
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen2-5-vl-7b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-14b": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.574,
          "reasoning": 1.434
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-14b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 14B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 1.147,
          "reasoning": 2.868
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 1.147,
          "reasoning": 2.868
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-32b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.072,
          "output": 0.287,
          "reasoning": 0.717
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-8b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 8B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-asr-flash": {
        "attachment": false,
        "cost": {
          "input": 0.032,
          "output": 0.032
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "qwen",
        "id": "qwen3-asr-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-09-08",
        "limit": {
          "context": 53248,
          "output": 4096
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-ASR Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-08",
        "temperature": false,
        "tool_call": false
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.216,
          "output": 0.861
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.861,
          "output": 3.441
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 480B-A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.574
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-flash",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Hosted Qwen coder for software agents, repo edits, and long-context code",
        "family": "qwen",
        "id": "qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 0.861,
          "output": 3.441
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 0.574
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.144,
          "output": 1.434
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B (Thinking)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-omni-flash": {
        "attachment": false,
        "cost": {
          "input": 0.058,
          "input_audio": 3.584,
          "output": 0.23,
          "output_audio": 7.168
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen3-omni-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen3-Omni Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-omni-flash-realtime": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "input_audio": 3.584,
          "output": 0.918,
          "output_audio": 7.168
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen3-omni-flash-realtime",
        "knowledge": "2024-04",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen3-Omni Flash Realtime",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-235b-a22b": {
        "attachment": false,
        "cost": {
          "input": 0.286705,
          "output": 1.14682,
          "reasoning": 2.867051
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-235b-a22b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL 235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.108,
          "output": 0.431,
          "reasoning": 1.076
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-30b-a3b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL 30B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-plus": {
        "attachment": false,
        "cost": {
          "input": 0.143353,
          "output": 1.433525,
          "reasoning": 4.300576
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-397b-a17b": {
        "attachment": false,
        "cost": {
          "input": 0.43,
          "output": 2.58,
          "reasoning": 2.58
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-397b-a17b",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.172,
          "output": 1.72,
          "reasoning": 1.72
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-plus": {
        "attachment": false,
        "cost": {
          "input": 0.573,
          "output": 3.44,
          "reasoning": 3.44
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_write": 0.234375,
          "input": 0.1875,
          "output": 1.125
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.132,
          "input": 1.32,
          "output": 7.9
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3.6-max-preview",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 245800,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 128000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "qwq-32b": {
        "attachment": false,
        "cost": {
          "input": 0.287,
          "output": 0.861
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwq-32b",
        "knowledge": "2024-04",
        "last_updated": "2024-12",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwQ 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12",
        "temperature": true,
        "tool_call": true
      },
      "qwq-plus": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "output": 0.574
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwq-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-03-05",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwQ Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-05",
        "temperature": true,
        "tool_call": true
      },
      "siliconflow/deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.18
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "siliconflow/deepseek-r1-0528",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "siliconflow/deepseek-r1-0528",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "siliconflow/deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "siliconflow/deepseek-v3-0324",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "siliconflow/deepseek-v3-0324",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "siliconflow/deepseek-v3.1-terminus": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "siliconflow/deepseek-v3.1-terminus",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "siliconflow/deepseek-v3.1-terminus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "siliconflow/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.42
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "siliconflow/deepseek-v3.2",
        "last_updated": "2025-12-03",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "siliconflow/deepseek-v3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "tongyi-intent-detect-v3": {
        "attachment": false,
        "cost": {
          "input": 0.058,
          "output": 0.144
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "yi",
        "id": "tongyi-intent-detect-v3",
        "knowledge": "2024-04",
        "last_updated": "2024-01",
        "limit": {
          "context": 8192,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tongyi Intent Detect V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Alibaba (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "alibaba-coding-plan": {
    "api": "https://coding-intl.dashscope.aliyuncs.com/v1",
    "doc": "https://www.alibabacloud.com/help/en/model-studio/coding-plan",
    "env": [
      "ALIBABA_CODING_PLAN_API_KEY"
    ],
    "id": "alibaba-coding-plan",
    "models": {
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "input": 196601,
          "output": 24576
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-next",
        "last_updated": "2026-02-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max-2026-01-23": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3-max-2026-01-23",
        "knowledge": "2025-04",
        "last_updated": "2026-01-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_write": 0.234375,
          "input": 0.1875,
          "output": 1.125
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Alibaba Coding Plan",
    "npm": "@ai-sdk/openai-compatible"
  },
  "alibaba-coding-plan-cn": {
    "api": "https://coding.dashscope.aliyuncs.com/v1",
    "doc": "https://help.aliyun.com/zh/model-studio/coding-plan",
    "env": [
      "ALIBABA_CODING_PLAN_API_KEY"
    ],
    "id": "alibaba-coding-plan-cn",
    "models": {
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 24576
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-next",
        "last_updated": "2026-02-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max-2026-01-23": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3-max-2026-01-23",
        "knowledge": "2025-04",
        "last_updated": "2026-01-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_write": 0.234375,
          "input": 0.1875,
          "output": 1.125
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Alibaba Coding Plan (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "alibaba-token-plan": {
    "api": "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
    "doc": "https://www.alibabacloud.com/help/en/model-studio/token-plan-overview",
    "env": [
      "ALIBABA_TOKEN_PLAN_API_KEY"
    ],
    "id": "alibaba-token-plan",
    "models": {
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "input": 196601,
          "output": 24576
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2025-01",
        "last_updated": "2025-12-05",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-12",
        "temperature": false,
        "tool_call": true
      },
      "qwen-image-2.0": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen-image-2.0",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen Image 2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": false
      },
      "qwen-image-2.0-pro": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen-image-2.0-pro",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen Image 2.0 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": false
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "wan2.7-image": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "wan2.7-image",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Wan2.7 Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": false
      },
      "wan2.7-image-pro": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "wan2.7-image-pro",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Wan2.7 Image Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Alibaba Token Plan",
    "npm": "@ai-sdk/openai-compatible"
  },
  "alibaba-token-plan-cn": {
    "api": "https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1",
    "doc": "https://www.alibabacloud.com/help/zh/model-studio/token-plan-overview",
    "env": [
      "ALIBABA_TOKEN_PLAN_API_KEY"
    ],
    "id": "alibaba-token-plan-cn",
    "models": {
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "input": 196601,
          "output": 24576
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2025-01",
        "last_updated": "2025-12-05",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-12",
        "temperature": false,
        "tool_call": true
      },
      "qwen-image-2.0": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen-image-2.0",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen Image 2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": false
      },
      "qwen-image-2.0-pro": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen-image-2.0-pro",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen Image 2.0 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": false
      },
      "qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "wan2.7-image": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "wan2.7-image",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Wan2.7 Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": false
      },
      "wan2.7-image-pro": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "wan2.7-image-pro",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 8192,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Wan2.7 Image Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Alibaba Token Plan (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "amazon-bedrock": {
    "doc": "https://docs.aws.amazon.com/bedrock/latest/userguide/models-supported.html",
    "env": [
      "AWS_ACCESS_KEY_ID",
      "AWS_SECRET_ACCESS_KEY",
      "AWS_REGION",
      "AWS_BEARER_TOKEN_BEDROCK"
    ],
    "id": "amazon-bedrock",
    "models": {
      "amazon.nova-2-lite-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.33,
          "output": 2.75
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "nova",
        "id": "amazon.nova-2-lite-v1:0",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova 2 Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "amazon.nova-lite-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.06,
          "output": 0.24
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-lite",
        "id": "amazon.nova-lite-v1:0",
        "knowledge": "2024-10",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "amazon.nova-micro-v1:0": {
        "attachment": false,
        "cost": {
          "cache_read": 0.00875,
          "input": 0.035,
          "output": 0.14
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-micro",
        "id": "amazon.nova-micro-v1:0",
        "knowledge": "2024-10",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Micro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "amazon.nova-pro-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 0.8,
          "output": 3.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova-pro",
        "id": "amazon.nova-pro-v1:0",
        "knowledge": "2024-10",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "anthropic.claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "anthropic.claude-haiku-4-5-20251001-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic.claude-haiku-4-5-20251001-v1:0",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-opus-4-1-20250805-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic.claude-opus-4-1-20250805-v1:0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-opus-4-5-20251101-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic.claude-opus-4-5-20251101-v1:0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-opus-4-6-v1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "anthropic.claude-opus-4-6-v1",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic.claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic.claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic.claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic.claude-sonnet-4-5-20250929-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic.claude-sonnet-4-5-20250929-v1:0",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic.claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic.claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "anthropic.claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "au.anthropic.claude-haiku-4-5-20251001-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "au.anthropic.claude-haiku-4-5-20251001-v1:0",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (AU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "au.anthropic.claude-opus-4-6-v1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.65,
          "cache_write": 20.625,
          "input": 16.5,
          "output": 82.5
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "au.anthropic.claude-opus-4-6-v1",
        "knowledge": "2025-05",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AU Anthropic Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "au.anthropic.claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "au.anthropic.claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 (AU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "au.anthropic.claude-sonnet-4-5-20250929-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "au.anthropic.claude-sonnet-4-5-20250929-v1:0",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (AU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "au.anthropic.claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.33,
          "cache_write": 4.125,
          "input": 3.3,
          "output": 16.5
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "au.anthropic.claude-sonnet-4-6",
        "knowledge": "2025-08",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AU Anthropic Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "au.anthropic.claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "au.anthropic.claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5 (AU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "deepseek.r1-v1:0": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek.r1-v1:0",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek.v3-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.58,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek.v3-v1:0",
        "knowledge": "2024-07",
        "last_updated": "2025-09-18",
        "limit": {
          "context": 163840,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek.v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.62,
          "output": 1.85
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek.v3.2",
        "knowledge": "2024-07",
        "last_updated": "2026-02-06",
        "limit": {
          "context": 163840,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "eu.anthropic.claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1.1,
          "cache_write": 13.75,
          "input": 11,
          "output": 55
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "eu.anthropic.claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "eu.anthropic.claude-haiku-4-5-20251001-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 1.375,
          "input": 1.1,
          "output": 5.5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "eu.anthropic.claude-haiku-4-5-20251001-v1:0",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "eu.anthropic.claude-opus-4-5-20251101-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "cache_write": 6.875,
          "input": 5.5,
          "output": 27.5
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "eu.anthropic.claude-opus-4-5-20251101-v1:0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "eu.anthropic.claude-opus-4-6-v1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "cache_write": 6.875,
          "input": 5.5,
          "output": 27.5
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "eu.anthropic.claude-opus-4-6-v1",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "eu.anthropic.claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "cache_write": 6.875,
          "input": 5.5,
          "output": 27.5
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "eu.anthropic.claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "eu.anthropic.claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "cache_write": 6.875,
          "input": 5.5,
          "output": 27.5
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "eu.anthropic.claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "eu.anthropic.claude-sonnet-4-5-20250929-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.33,
          "cache_write": 4.125,
          "input": 3.3,
          "output": 16.5
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "eu.anthropic.claude-sonnet-4-5-20250929-v1:0",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "eu.anthropic.claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.33,
          "cache_write": 4.125,
          "input": 3.3,
          "output": 16.5
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "eu.anthropic.claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "eu.anthropic.claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.22,
          "cache_write": 2.75,
          "input": 2.2,
          "output": 11
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "eu.anthropic.claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5 (EU)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "global.anthropic.claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "global.anthropic.claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "global.anthropic.claude-haiku-4-5-20251001-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "global.anthropic.claude-haiku-4-5-20251001-v1:0",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "global.anthropic.claude-opus-4-5-20251101-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "global.anthropic.claude-opus-4-5-20251101-v1:0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "global.anthropic.claude-opus-4-6-v1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "global.anthropic.claude-opus-4-6-v1",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "global.anthropic.claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "global.anthropic.claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "global.anthropic.claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "global.anthropic.claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "global.anthropic.claude-sonnet-4-5-20250929-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "global.anthropic.claude-sonnet-4-5-20250929-v1:0",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "global.anthropic.claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "global.anthropic.claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "global.anthropic.claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "global.anthropic.claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5 (Global)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "google.gemma-3-12b-it": {
        "attachment": false,
        "cost": {
          "input": 0.049999999999999996,
          "output": 0.09999999999999999
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google.gemma-3-12b-it",
        "knowledge": "2024-12",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 3 12B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google.gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0.2
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google.gemma-3-27b-it",
        "knowledge": "2025-07",
        "last_updated": "2025-07-27",
        "limit": {
          "context": 202752,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 3 27B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google.gemma-3-4b-it": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.08
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google.gemma-3-4b-it",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 4B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "jp.anthropic.claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "jp.anthropic.claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 (JP)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "jp.anthropic.claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "jp.anthropic.claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 (JP)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "jp.anthropic.claude-sonnet-4-5-20250929-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "jp.anthropic.claude-sonnet-4-5-20250929-v1:0",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (JP)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "jp.anthropic.claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "jp.anthropic.claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6 (JP)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "jp.anthropic.claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "jp.anthropic.claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5 (JP)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "meta.llama3-1-70b-instruct-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta.llama3-1-70b-instruct-v1:0",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta.llama3-1-8b-instruct-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.22
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta.llama3-1-8b-instruct-v1:0",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta.llama3-3-70b-instruct-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta.llama3-3-70b-instruct-v1:0",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta.llama4-maverick-17b-instruct-v1:0": {
        "attachment": true,
        "cost": {
          "input": 0.24,
          "output": 0.97
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta.llama4-maverick-17b-instruct-v1:0",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta.llama4-scout-17b-instruct-v1:0": {
        "attachment": true,
        "cost": {
          "input": 0.17,
          "output": 0.66
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta.llama4-scout-17b-instruct-v1:0",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 3500000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "minimax.minimax-m2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax.minimax-m2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 204608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax.minimax-m2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax.minimax-m2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax.minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax.minimax-m2.5",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 196608,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "mistral.devstral-2-123b": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral.devstral-2-123b",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2 123B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.magistral-small-2509": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral",
        "id": "mistral.magistral-small-2509",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 128000,
          "output": 40000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Small 1.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.ministral-3-14b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral.ministral-3-14b-instruct",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 14B 3.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.ministral-3-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral.ministral-3-3b-instruct",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.ministral-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral.ministral-3-8b-instruct",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.mistral-large-3-675b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral",
        "id": "mistral.mistral-large-3-675b-instruct",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.pixtral-large-2502-v1:0": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral vision-language model for image understanding and multimodal chat",
        "family": "mistral",
        "id": "mistral.pixtral-large-2502-v1:0",
        "last_updated": "2025-04-08",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral Large (25.02)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-08",
        "temperature": true,
        "tool_call": true
      },
      "mistral.voxtral-mini-3b-2507": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral",
        "id": "mistral.voxtral-mini-3b-2507",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "audio",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voxtral Mini 3B 2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral.voxtral-small-24b-2507": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.35
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral",
        "id": "mistral.voxtral-small-24b-2507",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voxtral Small 24B 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshot.kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshot.kimi-k2-thinking",
        "interleaved": true,
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262143,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai.kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi",
        "id": "moonshotai.kimi-k2.5",
        "interleaved": true,
        "last_updated": "2026-02-06",
        "limit": {
          "context": 262143,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia.nemotron-nano-12b-v2": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "family": "nemotron",
        "id": "nvidia.nemotron-nano-12b-v2",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron Nano 12B v2 VL BF16",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia.nemotron-nano-3-30b": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.24
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia.nemotron-nano-3-30b",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron Nano 3 30B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia.nemotron-nano-9b-v2": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.23
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia.nemotron-nano-9b-v2",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron Nano 9B v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia.nemotron-super-3-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.65
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia.nemotron-super-3-120b",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron 3 Super 120B A12B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai.gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 2.75,
          "output": 16.5
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "openai.gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "provider": {
          "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1",
          "npm": "@ai-sdk/amazon-bedrock/mantle",
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai.gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "input": 5.5,
          "output": 33
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai.gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "provider": {
          "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1",
          "npm": "@ai-sdk/amazon-bedrock/mantle",
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai.gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai.gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": false,
        "provider": {
          "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/v1",
          "npm": "@ai-sdk/amazon-bedrock/mantle",
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai.gpt-oss-120b-1:0": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai.gpt-oss-120b-1:0",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai.gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.3
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai.gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": false,
        "provider": {
          "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/v1",
          "npm": "@ai-sdk/amazon-bedrock/mantle",
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai.gpt-oss-20b-1:0": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.3
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai.gpt-oss-20b-1:0",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai.gpt-oss-safeguard-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "openai.gpt-oss-safeguard-120b",
        "last_updated": "2025-10-29",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS Safeguard 120B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai.gpt-oss-safeguard-20b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.2
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "openai.gpt-oss-safeguard-20b",
        "last_updated": "2025-10-29",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS Safeguard 20B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-235b-a22b-2507-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.88
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen.qwen3-235b-a22b-2507-v1:0",
        "knowledge": "2024-04",
        "last_updated": "2025-09-18",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-32b-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen.qwen3-32b-v1:0",
        "knowledge": "2024-04",
        "last_updated": "2025-09-18",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B (dense)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-coder-30b-a3b-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen.qwen3-coder-30b-a3b-v1:0",
        "knowledge": "2024-04",
        "last_updated": "2025-09-18",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B A3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-coder-480b-a35b-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 1.8
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen.qwen3-coder-480b-a35b-v1:0",
        "knowledge": "2024-04",
        "last_updated": "2025-09-18",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 1.8
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen.qwen3-coder-next",
        "last_updated": "2026-02-06",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-next-80b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 1.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen.qwen3-next-80b-a3b",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-Next-80B-A3B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen.qwen3-vl-235b-a22b": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen.qwen3-vl-235b-a22b",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-235B-A22B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "us.anthropic.claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "us.anthropic.claude-haiku-4-5-20251001-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "us.anthropic.claude-haiku-4-5-20251001-v1:0",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-opus-4-1-20250805-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "us.anthropic.claude-opus-4-1-20250805-v1:0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-opus-4-5-20251101-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "us.anthropic.claude-opus-4-5-20251101-v1:0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-opus-4-6-v1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "us.anthropic.claude-opus-4-6-v1",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "us.anthropic.claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "us.anthropic.claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "us.anthropic.claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "us.anthropic.claude-sonnet-4-5-20250929-v1:0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "us.anthropic.claude-sonnet-4-5-20250929-v1:0",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "us.anthropic.claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "us.anthropic.claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "us.anthropic.claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5 (US)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "us.deepseek.r1-v1:0": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving",
        "family": "deepseek-thinking",
        "id": "us.deepseek.r1-v1:0",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1 (US)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "us.meta.llama4-maverick-17b-instruct-v1:0": {
        "attachment": true,
        "cost": {
          "input": 0.24,
          "output": 0.97
        },
        "description": "Open multimodal Llama for strong reasoning with efficient everyday serving",
        "family": "llama",
        "id": "us.meta.llama4-maverick-17b-instruct-v1:0",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B Instruct (US)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "us.meta.llama4-scout-17b-instruct-v1:0": {
        "attachment": true,
        "cost": {
          "input": 0.17,
          "output": 0.66
        },
        "description": "Open Llama with long-context vision for efficient multimodal agents",
        "family": "llama",
        "id": "us.meta.llama4-scout-17b-instruct-v1:0",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 3500000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B Instruct (US)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "writer.palmyra-x4-v1:0": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "palmyra",
        "id": "writer.palmyra-x4-v1:0",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 122880,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Palmyra X4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "writer.palmyra-x5-v1:0": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 6
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "palmyra",
        "id": "writer.palmyra-x5-v1:0",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 1040000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Palmyra X5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "xai.grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "xai.grok-4.3",
        "last_updated": "2026-06-28",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "provider": {
          "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1",
          "npm": "@ai-sdk/amazon-bedrock/mantle",
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai.glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai.glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai.glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "zai.glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai.glm-5": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai.glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 202752,
          "output": 101376
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Amazon Bedrock",
    "npm": "@ai-sdk/amazon-bedrock"
  },
  "ambient": {
    "api": "https://api.ambient.xyz/v1",
    "doc": "https://ambient.xyz",
    "env": [
      "AMBIENT_API_KEY"
    ],
    "id": "ambient",
    "models": {
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "cache_write": 0,
          "input": 0.75,
          "output": 3.5
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Ambient",
    "npm": "@ai-sdk/openai-compatible"
  },
  "anthropic": {
    "doc": "https://docs.anthropic.com/en/docs/about-claude/models",
    "env": [
      "ANTHROPIC_API_KEY"
    ],
    "id": "anthropic",
    "models": {
      "claude-3-5-sonnet-20240620": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "claude-3-5-sonnet-20240620",
        "knowledge": "2024-04-30",
        "last_updated": "2024-06-20",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-20",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-5-sonnet-20241022": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "claude-3-5-sonnet-20241022",
        "knowledge": "2024-04-30",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.5 v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-7-sonnet-20250219": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "claude-3-7-sonnet-20250219",
        "knowledge": "2024-10-31",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-02-19",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-haiku-20240307": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-haiku",
        "id": "claude-3-haiku-20240307",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-opus-20240229": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-opus",
        "id": "claude-3-opus-20240229",
        "knowledge": "2023-08-31",
        "last_updated": "2024-02-29",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-02-29",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-sonnet-20240229": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 0.3,
          "input": 3,
          "output": 15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "claude-3-sonnet-20240229",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-04",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-04",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-0": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1-20250805",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5-20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1,
                "cache_write": 12.5,
                "input": 10,
                "output": 50
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-0": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-0",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Anthropic",
    "npm": "@ai-sdk/anthropic"
  },
  "anyapi": {
    "api": "https://api.anyapi.ai/v1",
    "doc": "https://docs.anyapi.ai",
    "env": [
      "ANYAPI_API_KEY"
    ],
    "id": "anyapi",
    "models": {
      "anthropic/claude-haiku-4-5": {
        "attachment": true,
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-6": {
        "attachment": true,
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-7": {
        "attachment": true,
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-5": {
        "attachment": true,
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-6": {
        "attachment": true,
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r-plus-08-2024": {
        "attachment": false,
        "description": "Cohere's RAG workhorse for long-context enterprise search and tool use",
        "family": "command-r",
        "id": "cohere/command-r-plus-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R+",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat": {
        "attachment": true,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-chat",
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Chat",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1": {
        "attachment": true,
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Reasoner",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-preview": {
        "attachment": true,
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/devstral-2512": {
        "attachment": false,
        "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes",
        "family": "devstral",
        "id": "mistralai/devstral-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-2512": {
        "attachment": true,
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "mistralai/mistral-large-2512",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3": {
        "attachment": true,
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": false,
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "perplexity/sonar-pro": {
        "attachment": true,
        "description": "Deeper Sonar search model with broader retrieval and stronger synthesis",
        "family": "sonar-pro",
        "id": "perplexity/sonar-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-reasoning-pro": {
        "attachment": true,
        "description": "Web-grounded Sonar for multi-step research questions that need cited reasoning",
        "family": "sonar-reasoning",
        "id": "perplexity/sonar-reasoning-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-4.3": {
        "attachment": true,
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "xai/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "AnyAPI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "atomic-chat": {
    "api": "http://127.0.0.1:1337/v1",
    "doc": "https://atomic.chat",
    "env": [
      "ATOMIC_CHAT_API_KEY"
    ],
    "id": "atomic-chat",
    "models": {
      "Meta-Llama-3_1-8B-Instruct-GGUF": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Meta-Llama-3_1-8B-Instruct-GGUF",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3.1 8B Instruct (GGUF)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "Qwen3_5-9B-MLX-4bit": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen3_5-9B-MLX-4bit",
        "last_updated": "2026-04-04",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 9B (MLX 4-bit)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-05",
        "temperature": true,
        "tool_call": true
      },
      "Qwen3_5-9B-Q4_K_M": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen3_5-9B-Q4_K_M",
        "last_updated": "2026-04-04",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 9B (Q4_K_M)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-05",
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-E4B-it-IQ4_XS": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-E4B-it-IQ4_XS",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 E4B Instruct (IQ4_XS)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": false
      },
      "gemma-4-E4B-it-MLX-4bit": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-E4B-it-MLX-4bit",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 E4B Instruct (MLX 4-bit)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Atomic Chat",
    "npm": "@ai-sdk/openai-compatible"
  },
  "auriko": {
    "api": "https://api.auriko.ai/v1",
    "doc": "https://docs.auriko.ai",
    "env": [
      "AURIKO_API_KEY"
    ],
    "id": "auriko",
    "models": {
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.8
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2-7": {
        "attachment": false,
        "cost": {
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax-m2-7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2-7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax-m2-7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "qwen-3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen-3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Auriko",
    "npm": "@ai-sdk/openai-compatible"
  },
  "azure": {
    "doc": "https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models",
    "env": [
      "AZURE_RESOURCE_NAME",
      "AZURE_API_KEY"
    ],
    "id": "azure",
    "models": {
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-31",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "knowledge": "2025-12-31",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "status": "beta",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "codestral-2501": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "codestral",
        "id": "codestral-2501",
        "knowledge": "2024-03",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral 25.01",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.375,
          "input": 1.5,
          "output": 6
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex-mini",
        "id": "codex-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-05-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codex Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-16",
        "temperature": false,
        "tool_call": true
      },
      "cohere-command-a": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "cohere-command-a",
        "knowledge": "2024-06-01",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": true
      },
      "cohere-command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere-command-r-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere-command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere's RAG workhorse for long-context enterprise search and tool use",
        "family": "command-r",
        "id": "cohere-command-r-plus-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R+",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere-embed-v-4-0": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere-embed-v-4-0",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-15",
        "temperature": false,
        "tool_call": false
      },
      "cohere-embed-v3-english": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere-embed-v3-english",
        "last_updated": "2023-11-07",
        "limit": {
          "context": 512,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v3 English",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-11-07",
        "temperature": false,
        "tool_call": false
      },
      "cohere-embed-v3-multilingual": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere-embed-v3-multilingual",
        "last_updated": "2023-11-07",
        "limit": {
          "context": 512,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v3 Multilingual",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-11-07",
        "temperature": false,
        "tool_call": false
      },
      "deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1",
        "knowledge": "2024-07",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-0528",
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1-0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 1.14,
          "output": 4.56
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3-0324",
        "knowledge": "2024-07",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3-0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "input": 0.56,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.1",
        "knowledge": "2024-07",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-21",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.58,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2-speciale": {
        "attachment": false,
        "cost": {
          "input": 0.58,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2-speciale",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2-Speciale",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "input": 0.19,
          "output": 0.51
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V4-Flash",
        "open_weights": true,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models",
          "npm": "@ai-sdk/openai-compatible",
          "shape": "completions"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "input": 1.74,
          "output": 3.48
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V4-Pro",
        "open_weights": true,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models",
          "npm": "@ai-sdk/openai-compatible",
          "shape": "completions"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-0125": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-0125",
        "knowledge": "2021-08",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 0125",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-0301": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-0301",
        "knowledge": "2021-08",
        "last_updated": "2023-03-01",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 0301",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-0613": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-0613",
        "knowledge": "2021-08",
        "last_updated": "2023-06-13",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 0613",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-06-13",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-1106": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-1106",
        "knowledge": "2021-08",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 1106",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-instruct",
        "knowledge": "2021-08",
        "last_updated": "2023-09-21",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-21",
        "temperature": true,
        "tool_call": false
      },
      "gpt-4": {
        "attachment": false,
        "cost": {
          "input": 60,
          "output": 120
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2023-03-14",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-32k": {
        "attachment": false,
        "cost": {
          "input": 60,
          "output": 120
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4-32k",
        "knowledge": "2023-11",
        "last_updated": "2023-03-14",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 32K",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-turbo-vision": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4-turbo-vision",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo Vision",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5-chat",
        "knowledge": "2024-10-24",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": false
      },
      "gpt-5-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "image",
            "audio"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt-codex",
        "id": "gpt-5.1-chat",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "image",
            "audio"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "image",
            "audio"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.2-chat",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.3-chat",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt-pro",
        "id": "gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-image-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 40
        },
        "description": "OpenAI image model for production generation, edits, and brand-safe visual workflows",
        "family": "gpt-image",
        "id": "gpt-image-1",
        "last_updated": "2025-04-24",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-Image-1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-24",
        "temperature": false,
        "tool_call": false
      },
      "gpt-image-1.5": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 32
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "gpt-image-1.5",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "GPT-Image-1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-25",
        "temperature": false,
        "tool_call": false
      },
      "gpt-image-2": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 30
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "gpt-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-Image-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-non-reasoning",
        "last_updated": "2025-06-27",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-27",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-reasoning",
        "last_updated": "2025-06-27",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-27",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20-non-reasoning": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-20-non-reasoning",
        "knowledge": "2025-09",
        "last_updated": "2026-04-08",
        "limit": {
          "context": 262000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-08",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20-reasoning": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-20-reasoning",
        "knowledge": "2025-09",
        "last_updated": "2026-04-08",
        "limit": {
          "context": 262000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-08",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-reasoning",
        "knowledge": "2025-07",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4 Fast (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": true,
        "knowledge": "2024-08",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": true,
        "knowledge": "2025-01",
        "last_updated": "2026-02-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models",
          "npm": "@ai-sdk/openai-compatible",
          "shape": "completions"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": false,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": true,
        "knowledge": "2025-01",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "provider": {
          "api": "https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models",
          "npm": "@ai-sdk/openai-compatible",
          "shape": "completions"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.37,
          "output": 0.37
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "llama-3.2-11b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-11B-Vision-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.2-90b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 2.04,
          "output": 2.04
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "llama-3.2-90b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-90B-Vision-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.71,
          "output": 0.71
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-maverick-17b-128e-instruct-fp8": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "llama-4-maverick-17b-128e-instruct-fp8",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-scout-17b-16e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.78
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "llama-4-scout-17b-16e-instruct",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama-3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2.68,
          "output": 3.54
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.61
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3-8B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama-3.1-405b-instruct": {
        "attachment": false,
        "cost": {
          "input": 5.33,
          "output": 16
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3.1-405b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-405B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2.68,
          "output": 3.54
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3.1-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.61
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3.1-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-8B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "ministral-3b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3b",
        "knowledge": "2024-03",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-2411",
        "knowledge": "2024-09",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 24.11",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-2505": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral-medium-2505",
        "knowledge": "2025-05",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistral-nemo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-2503": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-2503",
        "knowledge": "2024-09",
        "last_updated": "2025-03-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-01",
        "temperature": true,
        "tool_call": true
      },
      "model-router": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "model-router",
        "id": "model-router",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Model Router",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-19",
        "tool_call": true
      },
      "o1": {
        "attachment": false,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "temperature": false,
        "tool_call": true
      },
      "o1-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o1-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-09-12",
        "limit": {
          "context": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-12",
        "temperature": false,
        "tool_call": true
      },
      "o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "temperature": false,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "phi-3-medium-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-3-medium-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-medium-instruct (128k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-medium-4k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-3-medium-4k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 4096,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-medium-instruct (4k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-mini-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-mini-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-mini-instruct (128k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-mini-4k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-mini-4k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 4096,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-mini-instruct (4k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-small-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-small-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-small-instruct (128k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-small-8k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-small-8k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-small-instruct (8k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3.5-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3.5-mini-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-mini-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": false
      },
      "phi-3.5-moe-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.16,
          "output": 0.64
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-3.5-moe-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-MoE-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": false
      },
      "phi-4": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-4",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "phi-4-mini": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-4-mini",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "phi-4-mini-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-4-mini-reasoning",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini-reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "phi-4-multimodal": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "input_audio": 4,
          "output": 0.32
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "phi",
        "id": "phi-4-multimodal",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-multimodal",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "phi-4-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "phi-4-reasoning",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "phi-4-reasoning-plus": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "phi-4-reasoning-plus",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-reasoning-plus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "text-embedding-3-large": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-3-large",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8191,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "tool_call": false
      },
      "text-embedding-3-small": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-3-small",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8191,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-small",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "tool_call": false
      },
      "text-embedding-ada-002": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-ada-002",
        "last_updated": "2022-12-15",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-ada-002",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2022-12-15",
        "tool_call": false
      }
    },
    "name": "Azure",
    "npm": "@ai-sdk/azure"
  },
  "azure-cognitive-services": {
    "doc": "https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models",
    "env": [
      "AZURE_COGNITIVE_SERVICES_RESOURCE_NAME",
      "AZURE_COGNITIVE_SERVICES_API_KEY"
    ],
    "id": "azure-cognitive-services",
    "models": {
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-31",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "knowledge": "2025-12-31",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "codestral-2501": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "codestral",
        "id": "codestral-2501",
        "knowledge": "2024-03",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral 25.01",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.375,
          "input": 1.5,
          "output": 6
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex-mini",
        "id": "codex-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-05-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codex Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-16",
        "temperature": false,
        "tool_call": true
      },
      "cohere-command-a": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "cohere-command-a",
        "knowledge": "2024-06-01",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": true
      },
      "cohere-command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere-command-r-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere-command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere's RAG workhorse for long-context enterprise search and tool use",
        "family": "command-r",
        "id": "cohere-command-r-plus-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R+",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere-embed-v-4-0": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere-embed-v-4-0",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-15",
        "temperature": false,
        "tool_call": false
      },
      "cohere-embed-v3-english": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere-embed-v3-english",
        "last_updated": "2023-11-07",
        "limit": {
          "context": 512,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v3 English",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-11-07",
        "temperature": false,
        "tool_call": false
      },
      "cohere-embed-v3-multilingual": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere-embed-v3-multilingual",
        "last_updated": "2023-11-07",
        "limit": {
          "context": 512,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v3 Multilingual",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-11-07",
        "temperature": false,
        "tool_call": false
      },
      "deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1",
        "knowledge": "2024-07",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-0528",
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1-0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 1.14,
          "output": 4.56
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3-0324",
        "knowledge": "2024-07",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3-0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "input": 0.56,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.1",
        "knowledge": "2024-07",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-21",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.58,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2-speciale": {
        "attachment": false,
        "cost": {
          "input": 0.58,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2-speciale",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2-Speciale",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-0125": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-0125",
        "knowledge": "2021-08",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 0125",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-0301": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-0301",
        "knowledge": "2021-08",
        "last_updated": "2023-03-01",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 0301",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-0613": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-0613",
        "knowledge": "2021-08",
        "last_updated": "2023-06-13",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 0613",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-06-13",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-1106": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-1106",
        "knowledge": "2021-08",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 1106",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": false
      },
      "gpt-3.5-turbo-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo-instruct",
        "knowledge": "2021-08",
        "last_updated": "2023-09-21",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-21",
        "temperature": true,
        "tool_call": false
      },
      "gpt-4": {
        "attachment": false,
        "cost": {
          "input": 60,
          "output": 120
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2023-03-14",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-32k": {
        "attachment": false,
        "cost": {
          "input": 60,
          "output": 120
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4-32k",
        "knowledge": "2023-11",
        "last_updated": "2023-03-14",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 32K",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-turbo-vision": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4-turbo-vision",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo Vision",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5-chat",
        "knowledge": "2024-10-24",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": false
      },
      "gpt-5-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "image",
            "audio"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt-codex",
        "id": "gpt-5.1-chat",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "image",
            "audio"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text",
            "image",
            "audio"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.2-chat",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt-pro",
        "id": "gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "grok-4-fast-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-reasoning",
        "knowledge": "2025-07",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4 Fast (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": true,
        "knowledge": "2024-08",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": true,
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models",
          "npm": "@ai-sdk/openai-compatible",
          "shape": "completions"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": false,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": true,
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "provider": {
          "api": "https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models",
          "npm": "@ai-sdk/openai-compatible",
          "shape": "completions"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.37,
          "output": 0.37
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "llama-3.2-11b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-11B-Vision-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.2-90b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 2.04,
          "output": 2.04
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "llama-3.2-90b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-90B-Vision-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.71,
          "output": 0.71
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-maverick-17b-128e-instruct-fp8": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "llama-4-maverick-17b-128e-instruct-fp8",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-scout-17b-16e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.78
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "llama-4-scout-17b-16e-instruct",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama-3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2.68,
          "output": 3.54
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.61
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3-8B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama-3.1-405b-instruct": {
        "attachment": false,
        "cost": {
          "input": 5.33,
          "output": 16
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3.1-405b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-405B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2.68,
          "output": 3.54
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3.1-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.61
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama-3.1-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-8B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "ministral-3b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3b",
        "knowledge": "2024-03",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-2411",
        "knowledge": "2024-09",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 24.11",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-2505": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral-medium-2505",
        "knowledge": "2025-05",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistral-nemo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-2503": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-2503",
        "knowledge": "2024-09",
        "last_updated": "2025-03-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-01",
        "temperature": true,
        "tool_call": true
      },
      "model-router": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "model-router",
        "id": "model-router",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Model Router",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-19",
        "tool_call": true
      },
      "o1": {
        "attachment": false,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "temperature": false,
        "tool_call": true
      },
      "o1-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o1-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-09-12",
        "limit": {
          "context": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-12",
        "temperature": false,
        "tool_call": true
      },
      "o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "temperature": false,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "phi-3-medium-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-3-medium-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-medium-instruct (128k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-medium-4k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-3-medium-4k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 4096,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-medium-instruct (4k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-mini-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-mini-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-mini-instruct (128k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-mini-4k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-mini-4k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 4096,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-mini-instruct (4k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-small-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-small-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-small-instruct (128k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3-small-8k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3-small-8k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-small-instruct (8k)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": false
      },
      "phi-3.5-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-3.5-mini-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-mini-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": false
      },
      "phi-3.5-moe-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.16,
          "output": 0.64
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-3.5-moe-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-MoE-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": false
      },
      "phi-4": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "phi-4",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "phi-4-mini": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-4-mini",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "phi-4-mini-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "phi-4-mini-reasoning",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini-reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "phi-4-multimodal": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "input_audio": 4,
          "output": 0.32
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "phi",
        "id": "phi-4-multimodal",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-multimodal",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "phi-4-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "phi-4-reasoning",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "phi-4-reasoning-plus": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "phi-4-reasoning-plus",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-reasoning-plus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "text-embedding-3-large": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-3-large",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8191,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "tool_call": false
      },
      "text-embedding-3-small": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-3-small",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8191,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-small",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "tool_call": false
      },
      "text-embedding-ada-002": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-ada-002",
        "last_updated": "2022-12-15",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-ada-002",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2022-12-15",
        "tool_call": false
      }
    },
    "name": "Azure Cognitive Services",
    "npm": "@ai-sdk/azure"
  },
  "bailing": {
    "api": "https://api.tbox.cn/api/llm/v1/chat/completions",
    "doc": "https://alipaytbox.yuque.com/sxs0ba/ling/intro",
    "env": [
      "BAILING_API_TOKEN"
    ],
    "id": "bailing",
    "models": {
      "Ling-1T": {
        "attachment": false,
        "cost": {
          "input": 0.57,
          "output": 2.29
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "ling",
        "id": "Ling-1T",
        "knowledge": "2024-06",
        "last_updated": "2025-10",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling-1T",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10",
        "temperature": true,
        "tool_call": true
      },
      "Ring-1T": {
        "attachment": false,
        "cost": {
          "input": 0.57,
          "output": 2.29
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "ring",
        "id": "Ring-1T",
        "knowledge": "2024-06",
        "last_updated": "2025-10",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ring-1T",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Bailing",
    "npm": "@ai-sdk/openai-compatible"
  },
  "baseten": {
    "api": "https://inference.baseten.co/v1",
    "doc": "https://docs.baseten.co/inference/model-apis/overview",
    "env": [
      "BASETEN_API_KEY"
    ],
    "id": "baseten",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204000,
          "output": 204000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1",
        "last_updated": "2025-08-25",
        "limit": {
          "context": 164000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-25",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.145,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-12",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.7-Code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.7-Code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-04",
        "limit": {
          "context": 202800,
          "output": 202800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Ultra",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/Nemotron-120B-A12B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 0.75
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/Nemotron-120B-A12B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-02",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 202800,
          "output": 202800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Super",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.5
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "knowledge": "2025-08",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128072,
          "output": 128072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.95,
          "output": 3.15
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202800,
          "output": 202800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.3,
          "output": 4.3
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202800,
          "output": 202800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 202720,
          "output": 202720
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Baseten",
    "npm": "@ai-sdk/openai-compatible"
  },
  "berget": {
    "api": "https://api.berget.ai/v1",
    "doc": "https://api.berget.ai",
    "env": [
      "BERGET_API_KEY"
    ],
    "id": "berget",
    "models": {
      "google/gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.275,
          "output": 0.55
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "knowledge": "2025-12",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.99,
          "output": 0.99
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "knowledge": "2023-12",
        "last_updated": "2025-04-27",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Mistral-Medium-3.5-128B": {
        "attachment": true,
        "cost": {
          "input": 1.65,
          "output": 5.5
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/Mistral-Medium-3.5-128B",
        "knowledge": "2026-04",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5 128B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Mistral-Small-3.2-24B-Instruct-2506": {
        "attachment": false,
        "cost": {
          "input": 0.33,
          "output": 0.33
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistralai/Mistral-Small-3.2-24B-Instruct-2506",
        "knowledge": "2025-09",
        "last_updated": "2025-10-01",
        "limit": {
          "context": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2 24B Instruct 2506",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.83,
          "output": 3.85
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.83
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "knowledge": "2025-08",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS-120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.77,
          "output": 2.75
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.7",
        "knowledge": "2025-12",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "input": 1.54,
          "output": 4.84
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 524288,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Berget.AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "cerebras": {
    "doc": "https://inference-docs.cerebras.ai/models/overview",
    "env": [
      "CEREBRAS_API_KEY"
    ],
    "id": "cerebras",
    "models": {
      "gemma-4-31b": {
        "attachment": true,
        "cost": {
          "input": 0.99,
          "output": 1.49
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "gemma-4-31b",
        "last_updated": "2026-07-01",
        "limit": {
          "context": 131072,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 0.75
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "2026-06-10",
        "limit": {
          "context": 131072,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 2.25,
          "output": 2.75
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-glm-4.7",
        "last_updated": "2026-06-10",
        "limit": {
          "context": 131072,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.AI GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none"
            ]
          }
        ],
        "release_date": "2026-01-07",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Cerebras",
    "npm": "@ai-sdk/cerebras"
  },
  "chutes": {
    "api": "https://llm.chutes.ai/v1",
    "doc": "https://llm.chutes.ai/v1/models",
    "env": [
      "CHUTES_API_KEY"
    ],
    "id": "chutes",
    "models": {
      "MiniMaxAI/MiniMax-M2.5-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.14945,
          "input": 0.2989,
          "output": 1.1957
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-06-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking 2507 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-32B-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.052,
          "input": 0.104,
          "output": 0.416
        },
        "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-32B-TEE",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B-TEE": {
        "attachment": true,
        "cost": {
          "cache_read": 0.225,
          "input": 0.45,
          "output": 3
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-27B-TEE": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 0.3,
          "output": 2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-27B-TEE",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 1,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-21",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31B-turbo-TEE": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.12,
          "output": 0.37
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31B-turbo-TEE",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemma 4 31B turbo TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5-TEE": {
        "attachment": true,
        "cost": {
          "cache_read": 0.22,
          "input": 0.44,
          "output": 2
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6-TEE": {
        "attachment": true,
        "cost": {
          "cache_read": 0.33,
          "input": 0.66,
          "output": 3.5
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-12",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "unsloth/Mistral-Nemo-Instruct-2407-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01225,
          "input": 0.0245,
          "output": 0.0978
        },
        "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment",
        "family": "mistral-nemo",
        "id": "unsloth/Mistral-Nemo-Instruct-2407-TEE",
        "knowledge": "2024-07",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo Instruct 2407 TEE",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "temperature": true,
        "tool_call": false
      },
      "zai-org/GLM-5-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.475,
          "input": 0.95,
          "output": 2.55
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "zai-org/GLM-5-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.49,
          "input": 0.98,
          "output": 3.08
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2-TEE": {
        "attachment": false,
        "cost": {
          "cache_read": 0.7,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2-TEE",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 TEE",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Chutes",
    "npm": "@ai-sdk/openai-compatible"
  },
  "clarifai": {
    "api": "https://api.clarifai.com/v2/ext/openai/v1",
    "doc": "https://docs.clarifai.com/compute/inference/",
    "env": [
      "CLARIFAI_PAT"
    ],
    "id": "clarifai",
    "models": {
      "arcee_ai/AFM/models/trinity-mini": {
        "attachment": false,
        "cost": {
          "input": 0.045,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "trinity-mini",
        "id": "arcee_ai/AFM/models/trinity-mini",
        "knowledge": "2024-10",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Mini",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12",
        "temperature": true,
        "tool_call": true
      },
      "clarifai/main/models/mm-poly-8b": {
        "attachment": true,
        "cost": {
          "input": 0.658,
          "output": 1.11
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "mm-poly",
        "id": "clarifai/main/models/mm-poly-8b",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MM Poly 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-ai/deepseek-ocr/models/DeepSeek-OCR": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.7
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "family": "deepseek",
        "id": "deepseek-ai/deepseek-ocr/models/DeepSeek-OCR",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek OCR",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-20",
        "temperature": true,
        "tool_call": false
      },
      "minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5 High Throughput",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/completion/models/Ministral-3-14B-Reasoning-2512": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 1.7
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/completion/models/Ministral-3-14B-Reasoning-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 14B Reasoning 2512",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/completion/models/Ministral-3-3B-Reasoning-2512": {
        "attachment": true,
        "cost": {
          "input": 1.039,
          "output": 0.54825
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/completion/models/Ministral-3-3B-Reasoning-2512",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 3B Reasoning 2512",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/chat-completion/models/Kimi-K2_6": {
        "attachment": true,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/chat-completion/models/Kimi-K2_6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/chat-completion/models/gpt-oss-120b-high-throughput": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/chat-completion/models/gpt-oss-120b-high-throughput",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B High Throughput",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/chat-completion/models/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.045,
          "output": 0.18
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/chat-completion/models/gpt-oss-20b",
        "last_updated": "2025-12-12",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.11458,
          "output": 0.74812
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct",
        "knowledge": "2025-04",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.36,
          "output": 1.3
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Clarifai",
    "npm": "@ai-sdk/openai-compatible"
  },
  "claudinio": {
    "api": "https://api.claudin.io/v1",
    "doc": "https://claudin.io",
    "env": [
      "CLAUDINIO_API_KEY"
    ],
    "id": "claudinio",
    "models": {
      "claudinio": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 0.5,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "claudinio",
        "knowledge": "2026-05",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claudinio",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "tool_call": true
      },
      "claudius": {
        "attachment": true,
        "cost": {
          "cache_read": 0.9,
          "input": 3,
          "output": 8
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "claudius",
        "knowledge": "2026-05",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claudius",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "tool_call": true
      }
    },
    "name": "Claudinio",
    "npm": "@ai-sdk/openai-compatible"
  },
  "cloudferro-sherlock": {
    "api": "https://api-sherlock.cloudferro.com/openai/v1/",
    "doc": "https://docs.sherlock.cloudferro.com/",
    "env": [
      "CLOUDFERRO_SHERLOCK_API_KEY"
    ],
    "id": "cloudferro-sherlock",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "knowledge": "2026-01",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 196000,
          "input": 180000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 2.92,
          "output": 2.92
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "knowledge": "2024-10-09",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 70000,
          "output": 70000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 2.92,
          "output": 2.92
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "speakleash/Bielik-11B-v2.6-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.67,
          "output": 0.67
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "speakleash/Bielik-11B-v2.6-Instruct",
        "knowledge": "2025-03",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Bielik 11B v2.6 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "speakleash/Bielik-11B-v3.0-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.67,
          "output": 0.67
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "speakleash/Bielik-11B-v3.0-Instruct",
        "knowledge": "2025-03",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Bielik 11B v3.0 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "CloudFerro Sherlock",
    "npm": "@ai-sdk/openai-compatible"
  },
  "cloudflare-ai-gateway": {
    "doc": "https://developers.cloudflare.com/ai-gateway/",
    "env": [
      "CLOUDFLARE_API_TOKEN",
      "CLOUDFLARE_ACCOUNT_ID",
      "CLOUDFLARE_GATEWAY_ID"
    ],
    "id": "cloudflare-ai-gateway",
    "models": {
      "anthropic/claude-3-5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-3-5-haiku",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3.5 (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-3-haiku",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-3-opus",
        "knowledge": "2023-08-31",
        "last_updated": "2024-02-29",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-02-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 0.3,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-3-sonnet",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-04",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-04",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3.5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-3.5-haiku",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3.5 (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3.5-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-3.5-sonnet",
        "knowledge": "2024-04-30",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.5 v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "anthropic/claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-7",
        "knowledge": "2026-01",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-6",
        "interleaved": true,
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "provider": {
          "npm": "ai-gateway-provider"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-5",
        "interleaved": true,
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 1.25,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo",
        "knowledge": "2021-09-01",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "provider": {
          "npm": "ai-gateway-provider"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "provider": {
          "npm": "ai-gateway-provider"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "provider": {
          "npm": "ai-gateway-provider"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-pro": {
        "attachment": true,
        "cost": {
          "input": 20,
          "output": 80
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "openai/o3-pro",
        "knowledge": "2024-05",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.28,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "workers-ai/@cf/ai4bharat/indictrans2-en-indic-1B": {
        "attachment": false,
        "cost": {
          "input": 0.34,
          "output": 0.34
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "indictrans",
        "id": "workers-ai/@cf/ai4bharat/indictrans2-en-indic-1B",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "IndicTrans2 EN-Indic 1B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/aisingapore/gemma-sea-lion-v4-27b-it": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 0.56
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "workers-ai/@cf/aisingapore/gemma-sea-lion-v4-27b-it",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma SEA-LION v4 27B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/baai/bge-base-en-v1.5": {
        "attachment": false,
        "cost": {
          "input": 0.067,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "bge",
        "id": "workers-ai/@cf/baai/bge-base-en-v1.5",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE Base EN v1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/baai/bge-large-en-v1.5": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "bge",
        "id": "workers-ai/@cf/baai/bge-large-en-v1.5",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE Large EN v1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/baai/bge-m3": {
        "attachment": false,
        "cost": {
          "input": 0.012,
          "output": 0
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "bge",
        "id": "workers-ai/@cf/baai/bge-m3",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE M3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/baai/bge-reranker-base": {
        "attachment": false,
        "cost": {
          "input": 0.0031,
          "output": 0
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "bge",
        "id": "workers-ai/@cf/baai/bge-reranker-base",
        "last_updated": "2025-04-09",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE Reranker Base",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-09",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/baai/bge-small-en-v1.5": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "bge",
        "id": "workers-ai/@cf/baai/bge-small-en-v1.5",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE Small EN v1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/deepgram/aura-2-en": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "aura",
        "id": "workers-ai/@cf/deepgram/aura-2-en",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepgram Aura 2 (EN)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/deepgram/aura-2-es": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "aura",
        "id": "workers-ai/@cf/deepgram/aura-2-es",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepgram Aura 2 (ES)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/deepgram/nova-3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "nova",
        "id": "workers-ai/@cf/deepgram/nova-3",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepgram Nova 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 4.88
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "deepseek-thinking",
        "id": "workers-ai/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 32B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/facebook/bart-large-cnn": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "bart",
        "id": "workers-ai/@cf/facebook/bart-large-cnn",
        "last_updated": "2025-04-09",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BART Large CNN",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-09",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/google/gemma-3-12b-it": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 0.56
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "workers-ai/@cf/google/gemma-3-12b-it",
        "last_updated": "2025-04-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 12B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-11",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/huggingface/distilbert-sst-2-int8": {
        "attachment": false,
        "cost": {
          "input": 0.026,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "distilbert",
        "id": "workers-ai/@cf/huggingface/distilbert-sst-2-int8",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DistilBERT SST-2 INT8",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/ibm-granite/granite-4.0-h-micro": {
        "attachment": false,
        "cost": {
          "input": 0.017,
          "output": 0.11
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "granite",
        "id": "workers-ai/@cf/ibm-granite/granite-4.0-h-micro",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "IBM Granite 4.0 H Micro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-2-7b-chat-fp16": {
        "attachment": false,
        "cost": {
          "input": 0.56,
          "output": 6.67
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-2-7b-chat-fp16",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 2 7B Chat FP16",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 0.83
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3-8b-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 8B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3-8b-instruct-awq": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.27
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3-8b-instruct-awq",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 8B Instruct AWQ",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 0.8299999999999998
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.1-8b-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.1-8b-instruct-awq": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.27
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.1-8b-instruct-awq",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct AWQ",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.1-8b-instruct-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.29
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.1-8b-instruct-fp8",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct FP8",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.2-11b-vision-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.049,
          "output": 0.68
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.2-11b-vision-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11B Vision Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.2-1b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.027,
          "output": 0.2
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.2-1b-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 1B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.051,
          "output": 0.34
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.2-3b-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-3.3-70b-instruct-fp8-fast": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 2.25
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-3.3-70b-instruct-fp8-fast",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct FP8 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-4-scout-17b-16e-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.85
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-4-scout-17b-16e-instruct",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-16",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/llama-guard-3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.48,
          "output": 0.03
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "workers-ai/@cf/meta/llama-guard-3-8b",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama Guard 3 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/meta/m2m100-1.2b": {
        "attachment": false,
        "cost": {
          "input": 0.34,
          "output": 0.34
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "m2m",
        "id": "workers-ai/@cf/meta/m2m100-1.2b",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "M2M100 1.2B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/mistral/mistral-7b-instruct-v0.1": {
        "attachment": false,
        "cost": {
          "input": 0.11,
          "output": 0.19
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "workers-ai/@cf/mistral/mistral-7b-instruct-v0.1",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral 7B Instruct v0.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/mistralai/mistral-small-3.1-24b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 0.56
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "workers-ai/@cf/mistralai/mistral-small-3.1-24b-instruct",
        "last_updated": "2025-04-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.1 24B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-11",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "workers-ai/@cf/moonshotai/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "workers-ai/@cf/moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "workers-ai/@cf/moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "workers-ai/@cf/myshell-ai/melotts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "melotts",
        "id": "workers-ai/@cf/myshell-ai/melotts",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MyShell MeloTTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/nvidia/nemotron-3-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "workers-ai/@cf/nvidia/nemotron-3-120b-a12b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-11",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "workers-ai/@cf/openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 0.75
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "workers-ai/@cf/openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.3
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "workers-ai/@cf/openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/pfnet/plamo-embedding-1b": {
        "attachment": false,
        "cost": {
          "input": 0.019,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "plamo",
        "id": "workers-ai/@cf/pfnet/plamo-embedding-1b",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "PLaMo Embedding 1B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/pipecat-ai/smart-turn-v2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "smart-turn",
        "id": "workers-ai/@cf/pipecat-ai/smart-turn-v2",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pipecat Smart Turn v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/qwen/qwen2.5-coder-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.66,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "workers-ai/@cf/qwen/qwen2.5-coder-32b-instruct",
        "last_updated": "2025-04-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 Coder 32B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-11",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/qwen/qwen3-30b-a3b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.051,
          "output": 0.34
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "workers-ai/@cf/qwen/qwen3-30b-a3b-fp8",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B FP8",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/qwen/qwen3-embedding-0.6b": {
        "attachment": false,
        "cost": {
          "input": 0.012,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "workers-ai/@cf/qwen/qwen3-embedding-0.6b",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 0.6B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/qwen/qwq-32b": {
        "attachment": false,
        "cost": {
          "input": 0.66,
          "output": 1
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "workers-ai/@cf/qwen/qwq-32b",
        "last_updated": "2025-04-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwQ 32B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-11",
        "temperature": true,
        "tool_call": false
      },
      "workers-ai/@cf/zai-org/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "workers-ai/@cf/zai-org/glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Cloudflare AI Gateway",
    "npm": "ai-gateway-provider"
  },
  "cloudflare-workers-ai": {
    "api": "https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1",
    "doc": "https://developers.cloudflare.com/workers-ai/models/",
    "env": [
      "CLOUDFLARE_ACCOUNT_ID",
      "CLOUDFLARE_API_KEY"
    ],
    "id": "cloudflare-workers-ai",
    "models": {
      "@cf/aisingapore/gemma-sea-lion-v4-27b-it": {
        "attachment": false,
        "cost": {
          "input": 0.351,
          "output": 0.555
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "@cf/aisingapore/gemma-sea-lion-v4-27b-it",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma Sea Lion V4 27B It",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/deepseek-ai/deepseek-r1-distill-qwen-32b": {
        "attachment": false,
        "cost": {
          "input": 0.497,
          "output": 4.881
        },
        "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving",
        "family": "deepseek-thinking",
        "id": "@cf/deepseek-ai/deepseek-r1-distill-qwen-32b",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 80000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek R1 Distill Qwen 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "@cf/google/gemma-4-26b-a4b-it",
        "interleaved": true,
        "last_updated": "2026-04-02",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/ibm-granite/granite-4.0-h-micro": {
        "attachment": false,
        "cost": {
          "input": 0.017,
          "output": 0.112
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "granite",
        "id": "@cf/ibm-granite/granite-4.0-h-micro",
        "last_updated": "2025-10-07",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Granite 4.0 H Micro",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "@cf/meta/llama-3.1-8b-instruct-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.152,
          "output": 0.287
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "@cf/meta/llama-3.1-8b-instruct-fp8",
        "last_updated": "2024-07-25",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct fp8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/meta/llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.0485,
          "output": 0.676
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "@cf/meta/llama-3.2-11b-vision-instruct",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11B Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/meta/llama-3.2-1b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.027,
          "output": 0.201
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "@cf/meta/llama-3.2-1b-instruct",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 60000,
          "output": 60000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 1B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/meta/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.0509,
          "output": 0.335
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "@cf/meta/llama-3.2-3b-instruct",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 80000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/meta/llama-3.3-70b-instruct-fp8-fast": {
        "attachment": false,
        "cost": {
          "input": 0.293,
          "output": 2.253
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "@cf/meta/llama-3.3-70b-instruct-fp8-fast",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 24000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct fp8 Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "@cf/meta/llama-4-scout-17b-16e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.27,
          "output": 0.85
        },
        "description": "Open Llama with long-context vision for efficient multimodal agents",
        "family": "llama",
        "id": "@cf/meta/llama-4-scout-17b-16e-instruct",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 131000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "@cf/meta/llama-guard-3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.484,
          "output": 0.03
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "@cf/meta/llama-guard-3-8b",
        "last_updated": "2025-01-22",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama Guard 3 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/mistralai/mistral-small-3.1-24b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.351,
          "output": 0.555
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "@cf/mistralai/mistral-small-3.1-24b-instruct",
        "last_updated": "2025-03-18",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.1 24B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-18",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "@cf/moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "@cf/moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/moonshotai/kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "@cf/moonshotai/kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/nvidia/nemotron-3-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "@cf/nvidia/nemotron-3-120b-a12b",
        "interleaved": true,
        "last_updated": "2026-03-11",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 0.75
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "@cf/openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.3
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "@cf/openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/qwen/qwen2.5-coder-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.66,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "@cf/qwen/qwen2.5-coder-32b-instruct",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 Coder 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/qwen/qwen3-30b-a3b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.0509,
          "output": 0.335
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "@cf/qwen/qwen3-30b-a3b-fp8",
        "last_updated": "2025-04-30",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3b fp8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "@cf/qwen/qwq-32b": {
        "attachment": false,
        "cost": {
          "input": 0.66,
          "output": 1
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "@cf/qwen/qwq-32b",
        "last_updated": "2025-03-05",
        "limit": {
          "context": 24000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwq 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "@cf/zai-org/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.0605,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "@cf/zai-org/glm-4.7-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "@cf/zai-org/glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "@cf/zai-org/glm-5.2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Glm 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Cloudflare Workers AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "cohere": {
    "doc": "https://docs.cohere.com/docs/models",
    "env": [
      "COHERE_API_KEY"
    ],
    "id": "cohere",
    "models": {
      "c4ai-aya-expanse-32b": {
        "attachment": false,
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "id": "c4ai-aya-expanse-32b",
        "last_updated": "2024-10-24",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aya Expanse 32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-24",
        "temperature": true,
        "tool_call": false
      },
      "c4ai-aya-expanse-8b": {
        "attachment": false,
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "id": "c4ai-aya-expanse-8b",
        "last_updated": "2024-10-24",
        "limit": {
          "context": 8000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aya Expanse 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-24",
        "temperature": true,
        "tool_call": false
      },
      "c4ai-aya-vision-32b": {
        "attachment": true,
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "id": "c4ai-aya-vision-32b",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 16000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aya Vision 32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-04",
        "temperature": true,
        "tool_call": false
      },
      "c4ai-aya-vision-8b": {
        "attachment": true,
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "id": "c4ai-aya-vision-8b",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 16000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aya Vision 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-04",
        "temperature": true,
        "tool_call": false
      },
      "command-a-03-2025": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "command-a-03-2025",
        "knowledge": "2024-06-01",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": true
      },
      "command-a-plus-05-2026": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere's stronger command model for multilingual agents and enterprise workflows",
        "family": "command-a",
        "id": "command-a-plus-05-2026",
        "knowledge": "2025-04-01",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A Plus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "command-a-reasoning-08-2025": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "command-a-reasoning-08-2025",
        "knowledge": "2024-06-01",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A Reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-21",
        "temperature": true,
        "tool_call": true
      },
      "command-a-translate-08-2025": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "family": "command-a",
        "id": "command-a-translate-08-2025",
        "knowledge": "2024-06-01",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 8000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A Translate",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-28",
        "temperature": true,
        "tool_call": true
      },
      "command-a-vision-07-2025": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "command-a-vision-07-2025",
        "knowledge": "2024-06-01",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 128000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A Vision",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": false
      },
      "command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "command-r-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere's RAG workhorse for long-context enterprise search and tool use",
        "family": "command-r",
        "id": "command-r-plus-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R+",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "command-r7b-12-2024": {
        "attachment": false,
        "cost": {
          "input": 0.0375,
          "output": 0.15
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "command-r7b-12-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-12-02",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-02",
        "temperature": true,
        "tool_call": true
      },
      "command-r7b-arabic-02-2025": {
        "attachment": false,
        "cost": {
          "input": 0.0375,
          "output": 0.15
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "command-r7b-arabic-02-2025",
        "knowledge": "2024-06-01",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R7B Arabic",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-02-27",
        "temperature": true,
        "tool_call": true
      },
      "north-mini-code-1-0": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere coding model for practical software engineering and agentic edits",
        "family": "north",
        "id": "north-mini-code-1-0",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-09-23",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "North Mini Code",
        "open_weights": true,
        "provider": {
          "api": "https://api.cohere.ai/compatibility/v1",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Cohere",
    "npm": "@ai-sdk/cohere"
  },
  "cortecs": {
    "api": "https://api.cortecs.ai/v1",
    "doc": "https://api.cortecs.ai/v1/models",
    "env": [
      "CORTECS_API_KEY"
    ],
    "id": "cortecs",
    "models": {
      "claude-4-5-sonnet": {
        "attachment": true,
        "cost": {
          "input": 3.259,
          "output": 16.296
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-4-5-sonnet",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.5 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-4-6-sonnet": {
        "attachment": true,
        "cost": {
          "input": 3.59,
          "output": 17.92
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-4-6-sonnet",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "input": 1.09,
          "output": 5.43
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus4-5": {
        "attachment": true,
        "cost": {
          "input": 5.98,
          "output": 29.89
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus4-6": {
        "attachment": true,
        "cost": {
          "input": 5.98,
          "output": 29.89
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.56,
          "cache_write": 6.99,
          "input": 5.6,
          "output": 27.99
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.563,
          "cache_write": 7.049,
          "input": 5.64,
          "output": 28.198
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4": {
        "attachment": false,
        "cost": {
          "input": 3.307,
          "output": 16.536
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4",
        "knowledge": "2025-03",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "codestral-2508": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "mistral",
        "id": "codestral-2508",
        "knowledge": "2025-03",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral 2508",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.585,
          "output": 2.307
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-0528",
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 0.551,
          "output": 1.654
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3-0324",
        "knowledge": "2024-07",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.266,
          "output": 0.444
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.133,
          "output": 0.266
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 1.553,
          "output": 3.106
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "temperature": true,
        "tool_call": true
      },
      "devstral-2512": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "id": "devstral-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": false,
        "cost": {
          "input": 1.654,
          "output": 11.024
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5": {
        "attachment": false,
        "cost": {
          "input": 0.67,
          "output": 2.46
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 1.34
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-01",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 2.23
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 198000,
          "output": 198000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.53
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-4.7-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 203000,
          "output": 203000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-08",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "input": 1.08,
          "output": 3.44
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 202752
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.308,
          "cache_write": 1.544,
          "input": 1.235,
          "output": 4.118
        },
        "description": "Faster GLM-5 lane for coding agents that need lower latency",
        "family": "glm",
        "id": "glm-5-turbo",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.31,
          "output": 4.1
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-14",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.39,
          "input": 1.44,
          "output": 4.53
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.308,
          "cache_write": 1.544,
          "input": 1.235,
          "output": 4.118
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "glm-5v-turbo",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": false,
        "cost": {
          "input": 2.354,
          "output": 9.417
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-06",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 3,
          "output": 16.13
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "knowledge": "2024-01",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT Oss 120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "hermes-4-70b": {
        "attachment": false,
        "cost": {
          "input": 0.116,
          "output": 0.358
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "hermes-4-70b",
        "knowledge": "2023-12",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-26",
        "temperature": true,
        "tool_call": true
      },
      "intellect-3": {
        "attachment": true,
        "cost": {
          "input": 0.219,
          "output": 1.202
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "intellect-3",
        "knowledge": "2025-11",
        "last_updated": "2025-11-26",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "INTELLECT 3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-26",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.551,
          "output": 2.646
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-instruct",
        "knowledge": "2024-07",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.656,
          "output": 2.731
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-12",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.76
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.81,
          "output": 3.54
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-17",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.32,
          "input": 1.28,
          "output": 4.63
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "llama-3.1-405b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.1-405b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 405B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.089,
          "output": 0.275
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-maverick": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.151,
          "input": 0.124,
          "output": 0.603
        },
        "description": "Open multimodal Llama for strong reasoning with efficient everyday serving",
        "family": "llama",
        "id": "llama-4-maverick",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2": {
        "attachment": false,
        "cost": {
          "input": 0.39,
          "output": 1.57
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-11",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 400000,
          "output": 400000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.1": {
        "attachment": false,
        "cost": {
          "input": 0.34,
          "output": 1.34
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-12-23",
        "limit": {
          "context": 196000,
          "output": 196000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.32,
          "output": 1.18
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.7": {
        "attachment": false,
        "cost": {
          "input": 0.47,
          "output": 1.4
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 202752,
          "output": 196072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-m2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.089,
          "input": 0.355,
          "output": 1.775
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-2512": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "mistral-large-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "mixtral-8x7B-instruct-v0.1": {
        "attachment": false,
        "cost": {
          "input": 0.438,
          "output": 0.68
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "mixtral-8x7B-instruct-v0.1",
        "knowledge": "2023-09",
        "last_updated": "2023-12-11",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mixtral 8x7B Instruct v0.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2023-12-11",
        "temperature": true,
        "tool_call": false
      },
      "nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.266,
          "output": 0.799
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nemotron-3-super-120b-a12b",
        "knowledge": "2025-12",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super 120B A12B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "nova-pro-v1": {
        "attachment": false,
        "cost": {
          "input": 1.016,
          "output": 4.061
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova-pro",
        "id": "nova-pro-v1",
        "knowledge": "2024-04",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "output": 5000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Pro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "qwen-2.5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.062,
          "output": 0.231
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-2.5-72b-instruct",
        "knowledge": "2024-06",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 33000,
          "output": 33000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-19",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.062,
          "output": 0.408
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b-instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.099,
          "output": 0.33
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-32b",
        "knowledge": "2024-12",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.053,
          "output": 0.222
        },
        "description": "Smaller Qwen coder for efficient local agents and repo-level fixes",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B A3B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.441,
          "output": 1.984
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-01",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.158,
          "output": 0.84
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-next",
        "knowledge": "2025-04",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next 80B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.164,
          "output": 1.311
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-next-80b-a3b-thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-122b-a10b": {
        "attachment": false,
        "cost": {
          "input": 0.444,
          "output": 3.106
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-122b-a10b",
        "knowledge": "2026-01",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-24",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-397b-a17b": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen3.5-397b-a17b",
        "knowledge": "2026-01",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 250000,
          "output": 250000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Cortecs",
    "npm": "@ai-sdk/openai-compatible"
  },
  "crof": {
    "api": "https://crof.ai/v1",
    "doc": "https://crof.ai/docs",
    "env": [
      "CROF_API_KEY"
    ],
    "id": "crof",
    "models": {
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "input": 0.18,
          "output": 0.35
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-22",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003,
          "input": 0.12,
          "output": 0.21
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003,
          "input": 0.35,
          "output": 0.8
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro-lightning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.8,
          "output": 1.6
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro-lightning",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro Lightning",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0,
          "input": 0.25,
          "output": 1.1
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 202752,
          "output": 202752
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.008,
          "cache_write": 0,
          "input": 0.04,
          "output": 0.3
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm-flash",
        "id": "glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 0,
          "input": 0.48,
          "output": 1.9
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 202752
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 0,
          "input": 0.45,
          "output": 2.15
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 202752
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.5,
          "output": 2.2
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "greg-1-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.07,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "greg-1-mini",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 229376,
          "output": 229376
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Greg 1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": false
      },
      "greg-2-super": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "input": 1.5,
          "output": 5
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "greg-2-super",
        "last_updated": "2026-06-14",
        "limit": {
          "context": 229376,
          "output": 229376
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Greg 2 Super",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-14",
        "temperature": true,
        "tool_call": false
      },
      "greg-2-ultra": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 3,
          "output": 10
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "greg-2-ultra",
        "last_updated": "2026-06-14",
        "limit": {
          "context": 229376,
          "output": 229376
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Greg 2 Ultra",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-14",
        "temperature": true,
        "tool_call": false
      },
      "greg-rp": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "greg-rp",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 229376,
          "output": 229376
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Greg (Roleplay)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": false
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.07,
          "input": 0.35,
          "output": 1.7
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.5-lightning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5-lightning",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-06",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 (Lightning)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 1.99
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.55,
          "output": 2.25
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 0.4,
          "output": 0.8,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.375,
          "input": 0.11,
          "output": 0.95
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.07,
          "input": 0.35,
          "output": 1.75
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen3.5-397b-a17b",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-9b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.008,
          "input": 0.04,
          "output": 0.15
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-9b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-13",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.2,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.6-27b",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "CrofAI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "databricks": {
    "api": "https://${DATABRICKS_HOST}/ai-gateway/mlflow/v1",
    "doc": "https://docs.databricks.com/aws/en/machine-learning/foundation-models/",
    "env": [
      "DATABRICKS_HOST",
      "DATABRICKS_TOKEN"
    ],
    "id": "databricks",
    "models": {
      "databricks-claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "databricks-claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "databricks-claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "databricks-claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "databricks-claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "databricks-claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "databricks-claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "databricks-claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "databricks-claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "databricks-claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "databricks-claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "databricks-claude-sonnet-4",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "databricks-claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "databricks-claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "databricks-claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "databricks-claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "databricks-gemini-2-5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "databricks-gemini-2-5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gemini-2-5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "databricks-gemini-2-5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gemini-3-1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "databricks-gemini-3-1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gemini-3-1-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "databricks-gemini-3-1-pro",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gemini-3-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "databricks-gemini-3-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gemini-3-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "databricks-gemini-3-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "databricks-gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "databricks-gpt-5-1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "databricks-gpt-5-2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "databricks-gpt-5-4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.15,
                "input": 1.5,
                "output": 9
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt-mini",
        "id": "databricks-gpt-5-4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "databricks-gpt-5-4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "databricks-gpt-5-5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "databricks-gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "databricks-gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "databricks-gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.072,
          "output": 0.28
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "databricks-gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "databricks-gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "databricks-gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Databricks",
    "npm": "@ai-sdk/openai-compatible"
  },
  "deepinfra": {
    "doc": "https://deepinfra.com/models",
    "env": [
      "DEEPINFRA_API_KEY"
    ],
    "id": "deepinfra",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.15,
          "output": 1.15
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-06",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 66536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-35B-A3B": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.14,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-35B-A3B",
        "knowledge": "2025-01",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B": {
        "attachment": true,
        "cost": {
          "cache_read": 0.22,
          "input": 0.45,
          "output": 3
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B",
        "knowledge": "2025-01",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 397B A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.95
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-35B-A3B",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 262144,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "XiaomiMiMo/MiMo-V2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "context_over_200k": {
            "cache_read": 0.16,
            "input": 0.8,
            "output": 4
          },
          "input": 0.4,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.16,
              "input": 0.8,
              "output": 4,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "XiaomiMiMo/MiMo-V2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "XiaomiMiMo/MiMo-V2.5-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "XiaomiMiMo/MiMo-V2.5-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1-0528": {
        "attachment": false,
        "cost": {
          "cache_read": 0.35,
          "input": 0.5,
          "output": 2.15
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek-ai/DeepSeek-R1-0528",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1-0528",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 0.26,
          "output": 0.38
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 163840,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.2
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-ai/DeepSeek-V4-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 1.3,
          "output": 2.6
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26B-A4B-it": {
        "attachment": true,
        "cost": {
          "input": 0.07,
          "output": 0.34
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26B-A4B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.38
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.32
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "tool_call": true
      },
      "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "tool_call": false
      },
      "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 327680,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.07,
          "input": 0.45,
          "output": 2.25
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 0.75,
          "output": 3.5
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-04",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.7-Code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 0.74,
          "output": 3.5
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.7-Code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.039,
          "output": 0.19
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.14
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.43,
          "output": 1.74
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 1.75
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "zai-org/GLM-4.7-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 2.08
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-12",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.205,
          "input": 1.05,
          "output": 3.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.18,
          "input": 0.95,
          "output": 3
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1048576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Deep Infra",
    "npm": "@ai-sdk/deepinfra"
  },
  "deepseek": {
    "api": "https://api.deepseek.com",
    "doc": "https://api-docs.deepseek.com/quick_start/pricing",
    "env": [
      "DEEPSEEK_API_KEY"
    ],
    "id": "deepseek",
    "models": {
      "deepseek-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-chat",
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Chat",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-reasoner": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-reasoner",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Reasoner",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "DeepSeek",
    "npm": "@ai-sdk/openai-compatible"
  },
  "digitalocean": {
    "api": "https://inference.do-ai.run/v1",
    "doc": "https://docs.digitalocean.com/products/gradient-ai-platform/details/models/",
    "env": [
      "DIGITALOCEAN_ACCESS_TOKEN"
    ],
    "id": "digitalocean",
    "models": {
      "alibaba-qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.55
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba-qwen3-32b",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 131000,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "all-mini-lm-l6-v2": {
        "attachment": false,
        "cost": {
          "input": 0.009,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "all-mini-lm-l6-v2",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256,
          "output": 384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "All-MiniLM-L6-v2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2021-08-30",
        "temperature": false,
        "tool_call": false
      },
      "anthropic-claude-3-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-opus",
        "id": "anthropic-claude-3-opus",
        "knowledge": "2023-08",
        "last_updated": "2024-02-29",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3 Opus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-02-29",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-3.5-haiku": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-haiku",
        "id": "anthropic-claude-3.5-haiku",
        "knowledge": "2024-07",
        "last_updated": "2024-11-05",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-05",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-3.5-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "anthropic-claude-3.5-sonnet",
        "knowledge": "2024-04",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Sonnet",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-20",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-3.7-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "anthropic-claude-3.7-sonnet",
        "knowledge": "2024-11",
        "last_updated": "2025-02-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.7 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-02-24",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-4.1-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic-claude-4.1-opus",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-4.5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic-claude-4.5-haiku",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-4.5-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.3,
            "cache_write": 3.75,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.3,
              "cache_write": 3.75,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic-claude-4.5-sonnet",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-4.6-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.3,
            "cache_write": 3.75,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.3,
              "cache_write": 3.75,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic-claude-4.6-sonnet",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-fable-5": {
        "attachment": true,
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "anthropic-claude-fable-5",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic-claude-haiku-4.5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic-claude-opus-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic-claude-opus-4.5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "cache_write": 6.25,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 0.5,
              "cache_write": 6.25,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic-claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic-claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic-claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic-claude-opus-4.8",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": true
      },
      "anthropic-claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.3,
            "cache_write": 3.75,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.3,
              "cache_write": 3.75,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic-claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "arcee-trinity-large-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.25,
          "output": 0.9
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "trinity",
        "id": "arcee-trinity-large-thinking",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Large Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "bge-m3": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "bge",
        "id": "bge-m3",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 8192,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE M3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-01-30",
        "temperature": false,
        "tool_call": false
      },
      "bge-reranker-v2-m3": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "bge",
        "id": "bge-reranker-v2-m3",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 8192,
          "output": 1
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE Reranker v2 M3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-03-12",
        "temperature": false,
        "tool_call": false
      },
      "deepseek-3.2": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.6
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-3.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-4-flash": {
        "attachment": false,
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek",
        "id": "deepseek-4-flash",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek V4 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-27",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "input": 0.99,
          "output": 0.99
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-distill-llama-70b",
        "last_updated": "2025-01-30",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Llama 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2025-01-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3",
        "knowledge": "2024-07",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 163840,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-26",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "input": 1.74,
          "output": 3.48
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 393216
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "e5-large-v2": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "e5-large-v2",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 512,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "E5 Large v2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-05-19",
        "temperature": false,
        "tool_call": false
      },
      "fal-ai/elevenlabs/tts/multilingual-v2": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "elevenlabs",
        "id": "fal-ai/elevenlabs/tts/multilingual-v2",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "ElevenLabs Multilingual TTS v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-08-22",
        "temperature": false,
        "tool_call": false
      },
      "fal-ai/fast-sdxl": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "stable-diffusion",
        "id": "fal-ai/fast-sdxl",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Fast SDXL",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-07-26",
        "temperature": false,
        "tool_call": false
      },
      "fal-ai/flux/schnell": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "fal-ai/flux/schnell",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1 [schnell]",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-01",
        "temperature": false,
        "tool_call": false
      },
      "fal-ai/stable-audio-25/text-to-audio": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "fal-ai/stable-audio-25/text-to-audio",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Stable Audio 2.5 (Text-to-Audio)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-08",
        "temperature": false,
        "tool_call": false
      },
      "gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.5
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-31B-it",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-16",
        "limit": {
          "context": 202752,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-11",
        "tool_call": true
      },
      "gte-large-en-v1.5": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "gte-large-en-v1.5",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 8192,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GTE Large (v1.5)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-03-27",
        "temperature": false,
        "tool_call": false
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.7
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "llama-4-maverick": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 0.87
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "llama-4-maverick",
        "knowledge": "2024-08",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "llama3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.198,
          "output": 0.198
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama3-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 Instruct (8B)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "llama3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.65,
          "output": 0.65
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 Instruct 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax-m2.5",
        "id": "minimax-m2.5",
        "knowledge": "2025-08",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-12",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "ministral-3-8b-instruct-2512": {
        "attachment": true,
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3-8b-instruct-2512",
        "last_updated": "2025-12-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-15",
        "temperature": true,
        "tool_call": true
      },
      "mistral-3-14B": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral-3-14B",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 14B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-7b-instruct-v0.3": {
        "attachment": false,
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "mistral-7b-instruct-v0.3",
        "last_updated": "2024-05-22",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral 7B Instruct v0.3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-05-22",
        "temperature": true,
        "tool_call": true
      },
      "mistral-nemo-instruct-2407": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mistral",
        "id": "mistral-nemo-instruct-2407",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "multi-qa-mpnet-base-dot-v1": {
        "attachment": false,
        "cost": {
          "input": 0.009,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "multi-qa-mpnet-base-dot-v1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 512,
          "output": 768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Multi-QA-mpnet-base-dot-v1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2021-08-30",
        "temperature": false,
        "tool_call": false
      },
      "nemotron-3-nano-30b": {
        "attachment": false,
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nemotron-3-nano-30b",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Nano 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-nano-omni": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 0.9
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nemotron-3-nano-omni",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Nano 3 Omni",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-ultra-550b": {
        "attachment": false,
        "description": "Flagship Nemotron model for high-throughput reasoning and complex agents",
        "family": "nemotron",
        "id": "nemotron-3-ultra-550b",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-04",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-nano-12b-v2-vl": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "family": "nemotron",
        "id": "nemotron-nano-12b-v2-vl",
        "knowledge": "2024-10",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Nano 12B v2 VL",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "nvidia-nemotron-3-super-120b": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.65
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia-nemotron-3-super-120b",
        "knowledge": "2026-02",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron-3-Super-120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai-gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai-gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai-gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai-gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai-gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai-gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai-gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai-gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai-gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai-gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "openai-gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai-gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai-gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai-gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "openai-gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-image-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 40
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai-gpt-image-1",
        "last_updated": "2025-04-24",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-24",
        "temperature": false,
        "tool_call": false
      },
      "openai-gpt-image-1.5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "input": 5,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai-gpt-image-1.5",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-25",
        "temperature": false,
        "tool_call": false
      },
      "openai-gpt-image-2": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai-gpt-image-2",
        "last_updated": "2025-04-24",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-24",
        "temperature": false,
        "tool_call": false
      },
      "openai-gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.7
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai-gpt-oss-120b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-06",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.45
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai-gpt-oss-20b",
        "knowledge": "2024-06",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai-o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai-o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai-o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "qwen-2.5-14b-instruct": {
        "attachment": false,
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-2.5-14b-instruct",
        "knowledge": "2024-09",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 14B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-19",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 1.7
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Flash",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-embedding-0.6b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "qwen3-embedding-0.6b",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 8000,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 0.6B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-03",
        "status": "beta",
        "temperature": false,
        "tool_call": false
      },
      "qwen3-tts-voicedesign": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "qwen",
        "id": "qwen3-tts-voicedesign",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 32768,
          "output": 1
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Qwen3 TTS VoiceDesign",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "qwen3.5-397b-a17b": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 3.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen3.5",
        "id": "qwen3.5-397b-a17b",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 397B A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "stable-diffusion-3.5-large": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "stable-diffusion",
        "id": "stable-diffusion-3.5-large",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 256,
          "output": 1
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Stable Diffusion 3.5 Large",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": false,
        "tool_call": false
      },
      "wan2-2-t2v-a14b": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 0
        },
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "wan2-2-t2v-a14b",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 100,
          "output": 1
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan2.2-T2V-A14B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "DigitalOcean",
    "npm": "@ai-sdk/openai-compatible"
  },
  "dinference": {
    "api": "https://api.dinference.com/v1",
    "doc": "https://dinference.com",
    "env": [
      "DINFERENCE_API_KEY"
    ],
    "id": "dinference",
    "models": {
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 1.65
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "input": 0.75,
          "output": 2.4
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 3.89
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.0675,
          "output": 0.27
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.88
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "DInference",
    "npm": "@ai-sdk/openai-compatible"
  },
  "drun": {
    "api": "https://chat.d.run/v1",
    "doc": "https://www.d.run",
    "env": [
      "DRUN_API_KEY"
    ],
    "id": "drun",
    "models": {
      "public/deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.2
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "public/deepseek-r1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 131072,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "public/deepseek-v3": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "public/deepseek-v3",
        "knowledge": "2024-07",
        "last_updated": "2024-12-26",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-26",
        "temperature": true,
        "tool_call": true
      },
      "public/minimax-m25": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 1.16
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "public/minimax-m25",
        "interleaved": {
          "field": "reasoning_details"
        },
        "last_updated": "2025-03-01",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "D.Run (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "evroc": {
    "api": "https://models.think.evroc.com/v1",
    "doc": "https://docs.evroc.com/products/think/overview.html",
    "env": [
      "EVROC_API_KEY"
    ],
    "id": "evroc",
    "models": {
      "KBLab/kb-whisper-large": {
        "attachment": false,
        "cost": {
          "input": 0.0023,
          "output": 0.0023,
          "output_audio": 2.3
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "KBLab/kb-whisper-large",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 448,
          "output": 448
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KB Whisper",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "tool_call": false
      },
      "Qwen/Qwen3-Embedding-8B": {
        "attachment": false,
        "cost": {
          "input": 0.115,
          "output": 0.115
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "Qwen/Qwen3-Embedding-8B",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 40960,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "tool_call": false
      },
      "Qwen/Qwen3-Reranker-4B": {
        "attachment": false,
        "cost": {
          "input": 0.0575,
          "output": 0
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "qwen",
        "id": "Qwen/Qwen3-Reranker-4B",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Reranker 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "tool_call": false
      },
      "Qwen/Qwen3-VL-30B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "output": 0.92
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-30B-A3B-Instruct",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 100000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 30B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B-FP8": {
        "attachment": true,
        "cost": {
          "input": 0.345,
          "output": 1.38
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-35B-A3B-FP8",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "evroc/roc": {
        "attachment": true,
        "cost": {
          "input": 2.875,
          "output": 11.516
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "evroc/roc",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01",
        "last_updated": "2026-06-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "roc",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26B-A4B-it": {
        "attachment": true,
        "cost": {
          "input": 0.144,
          "output": 0.575
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26B-A4B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "intfloat/multilingual-e5-large-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.114,
          "output": 0.114
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "intfloat/multilingual-e5-large-instruct",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 512,
          "output": 512
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "E5 Multi-Lingual Large Embeddings 0.6B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-01",
        "tool_call": false
      },
      "mistralai/Mistral-Medium-3.5-128B": {
        "attachment": true,
        "cost": {
          "input": 1.725,
          "output": 6.9
        },
        "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools",
        "family": "mistral-medium",
        "id": "mistralai/Mistral-Medium-3.5-128B",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Voxtral-Small-24B-2507": {
        "attachment": false,
        "cost": {
          "input": 0.0023,
          "output": 0.0023,
          "output_audio": 2.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "voxtral",
        "id": "mistralai/Voxtral-Small-24B-2507",
        "last_updated": "2025-03-01",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "audio",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voxtral Small 24B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-01",
        "tool_call": false
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "input": 1.4375,
          "output": 5.75
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/Llama-3.3-70B-Instruct-FP8": {
        "attachment": true,
        "cost": {
          "input": 1.15,
          "output": 1.15
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "nvidia/Llama-3.3-70B-Instruct-FP8",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "output": 0.92
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "tool_call": true
      },
      "openai/whisper-large-v3": {
        "attachment": false,
        "cost": {
          "input": 0.0023,
          "output": 0.0023,
          "output_audio": 2.3
        },
        "description": "Open Whisper checkpoint for robust multilingual transcription and captioning",
        "family": "whisper",
        "id": "openai/whisper-large-v3",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 448,
          "output": 4096
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper 3 Large",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "tool_call": false
      },
      "openai/whisper-large-v3-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.0023,
          "output": 0.0023,
          "output_audio": 2.3
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "openai/whisper-large-v3-turbo",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 448,
          "output": 448
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper Large v3 Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "tool_call": false
      }
    },
    "name": "evroc",
    "npm": "@ai-sdk/openai-compatible"
  },
  "fastrouter": {
    "api": "https://go.fastrouter.ai/api/v1",
    "doc": "https://fastrouter.ai/models",
    "env": [
      "FASTROUTER_API_KEY"
    ],
    "id": "fastrouter",
    "models": {
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "bytedance/seedance-2": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "seed",
        "id": "bytedance/seedance-2",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 4096,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Seedance 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-01",
        "temperature": false,
        "tool_call": false
      },
      "deepseek-ai/deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.14
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/deepseek-r1-distill-llama-70b",
        "knowledge": "2024-10",
        "last_updated": "2025-01-23",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Llama 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-23",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "input": 1.74,
          "output": 3.48
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0375,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.31,
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-image-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12
        },
        "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 65536,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-image-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini-flash",
        "id": "google/gemini-3.1-flash-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.38
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/imagen-4.0-fast": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4.0-fast",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen 4 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": false,
        "tool_call": false
      },
      "google/imagen-4.0-ultra": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4.0-ultra",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen 4 Ultra",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": false,
        "tool_call": false
      },
      "google/veo3.1": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo3.1",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 400000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-01",
        "temperature": false,
        "tool_call": false
      },
      "google/veo3.1-fast": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo3.1-fast",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 400000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.1 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-01",
        "temperature": false,
        "tool_call": false
      },
      "google/veo3.1-lite": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo3.1-lite",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 400000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.1 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-01",
        "temperature": false,
        "tool_call": false
      },
      "leonardo-ai/lucid-origin": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "lucid",
        "id": "leonardo-ai/lucid-origin",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 4096,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Lucid Origin",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": false,
        "tool_call": false
      },
      "leonardo-ai/lucid-realism": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "lucid",
        "id": "leonardo-ai/lucid-realism",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 4096,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Lucid Realism",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": false,
        "tool_call": false
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax/minimax-m2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2",
        "knowledge": "2024-10",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 3.5
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-10-01",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-10-01",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-10-01",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 30
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "openai/gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-image-2": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai/gpt-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-realtime-1.5": {
        "attachment": true,
        "cost": {
          "input": 4,
          "output": 16
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "openai/gpt-realtime-1.5",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "audio",
            "image"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "GPT Realtime 1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 66536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "sarvam/sarvam-105b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.16
        },
        "description": "Flagship Indian-language reasoning model for enterprise multilingual applications",
        "family": "sarvam",
        "id": "sarvam/sarvam-105b",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sarvam 105B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": true
      },
      "sarvam/sarvam-30b": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.1
        },
        "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work",
        "family": "sarvam",
        "id": "sarvam/sarvam-30b",
        "last_updated": "2026-02-18",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sarvam 30B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-18",
        "temperature": true,
        "tool_call": true
      },
      "wanx/wan-v2-6": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "wanx/wan-v2-6",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 400000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan 2.6",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": false,
        "tool_call": false
      },
      "x-ai/grok-4": {
        "attachment": false,
        "cost": {
          "cache_read": 0.75,
          "cache_write": 15,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "x-ai/grok-4",
        "knowledge": "2025-07",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-09",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.3": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 2.5
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "x-ai/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-build-0.1": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 2
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "x-ai/grok-build-0.1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5": {
        "attachment": false,
        "cost": {
          "input": 0.95,
          "output": 3.15
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "z-ai/glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.1": {
        "attachment": false,
        "cost": {
          "input": 1.05,
          "output": 3.5
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "z-ai/glm-5.1",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "FastRouter",
    "npm": "@ai-sdk/openai-compatible"
  },
  "fireworks-ai": {
    "api": "https://api.fireworks.ai/inference/v1/",
    "doc": "https://fireworks.ai/docs/",
    "env": [
      "FIREWORKS_API_KEY"
    ],
    "id": "fireworks-ai",
    "models": {
      "accounts/fireworks/models/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "accounts/fireworks/models/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.145,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "accounts/fireworks/models/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/glm-5p1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "accounts/fireworks/models/glm-5p1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 202800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/glm-5p2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "accounts/fireworks/models/glm-5p2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-16",
        "limit": {
          "context": 1048575,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-16",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "accounts/fireworks/models/gpt-oss-120b",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.035,
          "input": 0.07,
          "output": 0.3
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "accounts/fireworks/models/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/kimi-k2p6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "accounts/fireworks/models/kimi-k2p6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/kimi-k2p7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi coding model for software agents, refactors, and repository reasoning",
        "family": "kimi-k2",
        "id": "accounts/fireworks/models/kimi-k2p7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-16",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-12",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/minimax-m2p7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "accounts/fireworks/models/minimax-m2p7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-12",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-12",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "accounts/fireworks/models/minimax-m3",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-12",
        "limit": {
          "context": 512000,
          "output": 512000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/models/qwen3p7-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "accounts/fireworks/models/qwen3p7-plus",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-12",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/routers/glm-5p1-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.52,
          "input": 2.8,
          "output": 8.8
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "accounts/fireworks/routers/glm-5p1-fast",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 202800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1 Fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/routers/glm-5p2-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.21,
          "input": 2.1,
          "output": 6.6
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "accounts/fireworks/routers/glm-5p2-fast",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-26",
        "limit": {
          "context": 1048575,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-26",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/routers/kimi-k2p6-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "input": 2,
          "output": 8
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "accounts/fireworks/routers/kimi-k2p6-fast",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-05",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 Fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/routers/kimi-k2p6-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "input": 2,
          "output": 8
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "accounts/fireworks/routers/kimi-k2p6-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 Turbo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "temperature": true,
        "tool_call": true
      },
      "accounts/fireworks/routers/kimi-k2p7-code-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.38,
          "input": 1.9,
          "output": 8
        },
        "description": "Kimi coding model for software agents, refactors, and repository reasoning",
        "family": "kimi-k2",
        "id": "accounts/fireworks/routers/kimi-k2p7-code-fast",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-16",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code Fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-12",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Fireworks AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "freemodel": {
    "api": "https://cc.freemodel.dev/v1",
    "doc": "https://freemodel.dev",
    "env": [
      "FREEMODEL_API_KEY"
    ],
    "id": "freemodel",
    "models": {
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "cache_write": 1.75,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "provider": {
          "api": "https://api.freemodel.dev/v1",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 2.5,
          "input": 2.5,
          "output": 15
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "provider": {
          "api": "https://api.freemodel.dev/v1",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "cache_write": 0.75,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "provider": {
          "api": "https://api.freemodel.dev/v1",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 5,
          "input": 5,
          "output": 30
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "provider": {
          "api": "https://api.freemodel.dev/v1",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "FreeModel",
    "npm": "@ai-sdk/anthropic"
  },
  "friendli": {
    "api": "https://api.friendli.ai/serverless/v1",
    "doc": "https://friendli.ai/docs/guides/serverless_endpoints/introduction",
    "env": [
      "FRIENDLI_TOKEN"
    ],
    "id": "friendli",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "input": 0.5,
          "output": 1.5
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.4
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 202752
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Friendli",
    "npm": "@ai-sdk/openai-compatible"
  },
  "frogbot": {
    "api": "https://app.frogbot.ai/api/v1",
    "doc": "https://docs.frogbot.ai",
    "env": [
      "FROGBOT_API_KEY"
    ],
    "id": "frogbot",
    "models": {
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.14,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek",
        "id": "deepseek-v4-pro",
        "knowledge": "2026-01",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek v4 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-24",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-07-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.31,
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3-1-pro-preview",
        "knowledge": "2026-01",
        "last_updated": "2026-02-18",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "gpt-5-3-codex",
        "knowledge": "2026-01-31",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-15",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-4-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "gpt-5-5",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "temperature": false,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "1970-01-01",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "1970-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-20b",
        "last_updated": "1970-01-01",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "1970-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-non-reasoning",
        "knowledge": "2025-11",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 2000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-reasoning",
        "knowledge": "2025-11",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 2000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-3",
        "knowledge": "2024-11",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-code-fast-1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-code-fast-1",
        "knowledge": "2023-10",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-28",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "kimi-k2-6",
        "last_updated": "1970-01-01",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "1970-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "kimi-k2.5",
        "last_updated": "1970-01-01",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "1970-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2-5",
        "knowledge": "2024-09",
        "last_updated": "2025-02-22",
        "limit": {
          "context": 192000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-15",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2-7",
        "knowledge": "2024-09",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 192000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "qwen-3-6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.5,
          "output": 3
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-3-6-plus",
        "last_updated": "2026-04-03",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "zai-glm-5-1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-glm-5-1",
        "knowledge": "2024-10",
        "last_updated": "2025-02-22",
        "limit": {
          "context": 198000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.AI GLM-5.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "FrogBot",
    "npm": "@ai-sdk/openai-compatible"
  },
  "github-copilot": {
    "api": "https://api.githubcopilot.com",
    "doc": "https://docs.github.com/en/copilot",
    "env": [
      "GITHUB_TOKEN"
    ],
    "id": "github-copilot",
    "models": {
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "claude-haiku-4.5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "input": 136000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4.5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "input": 168000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 200000,
          "input": 168000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 200000,
          "input": 168000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1,
                "cache_write": 12.5,
                "input": 10,
                "output": 50
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 200000,
          "input": 168000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 216000,
          "input": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4.5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "input": 168000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 200000,
          "input": 168000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 32000,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 32000,
            "min": 256,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 200000,
          "input": 136000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 32000,
            "min": 256,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "input_audio": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 200000,
          "input": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 24000,
            "min": 256,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 264000,
          "input": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 256000,
          "input": 224000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "mai-code-1-flash-picker": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Microsoft coding model built for fast, efficient assistance in everyday developer workflows",
        "family": "mai",
        "id": "mai-code-1-flash-picker",
        "knowledge": "2025-12",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 256000,
          "input": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MAI-Code-1-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "GitHub Copilot",
    "npm": "@ai-sdk/openai-compatible"
  },
  "github-models": {
    "api": "https://models.github.ai/inference",
    "doc": "https://docs.github.com/en/github-models",
    "env": [
      "GITHUB_TOKEN"
    ],
    "id": "github-models",
    "models": {
      "ai21-labs/ai21-jamba-1.5-large": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "jamba",
        "id": "ai21-labs/ai21-jamba-1.5-large",
        "knowledge": "2024-03",
        "last_updated": "2024-08-29",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AI21 Jamba 1.5 Large",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-08-29",
        "temperature": true,
        "tool_call": true
      },
      "ai21-labs/ai21-jamba-1.5-mini": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "jamba",
        "id": "ai21-labs/ai21-jamba-1.5-mini",
        "knowledge": "2024-03",
        "last_updated": "2024-08-29",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AI21 Jamba 1.5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-08-29",
        "temperature": true,
        "tool_call": true
      },
      "cohere/cohere-command-a": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "cohere/cohere-command-a",
        "knowledge": "2024-03",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command A",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "cohere/cohere-command-r": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/cohere-command-r",
        "knowledge": "2024-03",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command R",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-03-11",
        "temperature": true,
        "tool_call": true
      },
      "cohere/cohere-command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/cohere-command-r-08-2024",
        "knowledge": "2024-03",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command R 08-2024",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "temperature": true,
        "tool_call": true
      },
      "cohere/cohere-command-r-plus": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/cohere-command-r-plus",
        "knowledge": "2024-03",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command R+",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-04",
        "temperature": true,
        "tool_call": true
      },
      "cohere/cohere-command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/cohere-command-r-plus-08-2024",
        "knowledge": "2024-03",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command R+ 08-2024",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "temperature": true,
        "tool_call": true
      },
      "core42/jais-30b-chat": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "jais",
        "id": "core42/jais-30b-chat",
        "knowledge": "2023-03",
        "last_updated": "2023-08-30",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "JAIS 30b Chat",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2023-08-30",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1",
        "knowledge": "2024-06",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-0528",
        "knowledge": "2024-06",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1-0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3-0324",
        "knowledge": "2024-06",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3-0324",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-24",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-11b-vision-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta/llama-3.2-11b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-11B-Vision-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-90b-vision-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta/llama-3.2-90b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-90B-Vision-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-4-maverick-17b-128e-instruct-fp8": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta/llama-4-maverick-17b-128e-instruct-fp8",
        "knowledge": "2024-12",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-4-scout-17b-16e-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta/llama-4-scout-17b-16e-instruct",
        "knowledge": "2024-12",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": true,
        "tool_call": true
      },
      "meta/meta-llama-3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/meta-llama-3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3-70B-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": true
      },
      "meta/meta-llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/meta-llama-3-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3-8B-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": true
      },
      "meta/meta-llama-3.1-405b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/meta-llama-3.1-405b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-405B-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta/meta-llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/meta-llama-3.1-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-70B-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta/meta-llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/meta-llama-3.1-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-8B-Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/mai-ds-r1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "mai",
        "id": "microsoft/mai-ds-r1",
        "knowledge": "2024-06",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MAI-DS-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3-medium-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "microsoft/phi-3-medium-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-medium instruct (128k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3-medium-4k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "microsoft/phi-3-medium-4k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 4096,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-medium instruct (4k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3-mini-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-3-mini-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-mini instruct (128k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3-mini-4k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-3-mini-4k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 4096,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-mini instruct (4k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3-small-128k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-3-small-128k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-small instruct (128k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3-small-8k-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-3-small-8k-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-04-23",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3-small instruct (8k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-04-23",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3.5-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-3.5-mini-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-mini instruct (128k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3.5-moe-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "microsoft/phi-3.5-moe-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-MoE instruct (128k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-3.5-vision-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "phi",
        "id": "microsoft/phi-3.5-vision-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-08-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-3.5-vision instruct (128k)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-08-20",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-4": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "microsoft/phi-4",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-4-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-4-mini-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini-instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-4-mini-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-4-mini-reasoning",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini-reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-4-multimodal-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "phi",
        "id": "microsoft/phi-4-multimodal-instruct",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-multimodal-instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-4-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "phi",
        "id": "microsoft/phi-4-reasoning",
        "knowledge": "2023-10",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-Reasoning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "mistral-ai/codestral-2501": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "codestral",
        "id": "mistral-ai/codestral-2501",
        "knowledge": "2024-03",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral 25.01",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-ai/ministral-3b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral-ai/ministral-3b",
        "knowledge": "2024-03",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "mistral-ai/mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-ai/mistral-large-2411",
        "knowledge": "2024-09",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 24.11",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-ai/mistral-medium-2505": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral-ai/mistral-medium-2505",
        "knowledge": "2024-09",
        "last_updated": "2025-05-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3 (25.05)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-ai/mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistral-ai/mistral-nemo",
        "knowledge": "2024-03",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "mistral-ai/mistral-small-2503": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-ai/mistral-small-2503",
        "knowledge": "2024-09",
        "last_updated": "2025-03-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-01",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1-nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "knowledge": "2023-10",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2023-10",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/o1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "knowledge": "2023-10",
        "last_updated": "2024-12-17",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-12",
        "temperature": false,
        "tool_call": false
      },
      "openai/o1-mini": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o1-mini",
        "knowledge": "2023-10",
        "last_updated": "2024-12-17",
        "limit": {
          "context": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o1-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-12",
        "temperature": false,
        "tool_call": false
      },
      "openai/o1-preview": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1-preview",
        "knowledge": "2023-10",
        "last_updated": "2024-09-12",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o1-preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-09-12",
        "temperature": false,
        "tool_call": false
      },
      "openai/o3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-04",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": false,
        "tool_call": false
      },
      "openai/o3-mini": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": false,
        "tool_call": false
      },
      "openai/o4-mini": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": false,
        "tool_call": false
      },
      "xai/grok-3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-3",
        "knowledge": "2024-10",
        "last_updated": "2024-12-09",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-09",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-3-mini": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-3-mini",
        "knowledge": "2024-10",
        "last_updated": "2024-12-09",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 3 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-09",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "GitHub Models",
    "npm": "@ai-sdk/openai-compatible"
  },
  "gitlab": {
    "doc": "https://docs.gitlab.com/user/duo_agent_platform/",
    "env": [
      "GITLAB_TOKEN"
    ],
    "id": "gitlab",
    "models": {
      "duo-chat-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "duo-chat-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Fable 5)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-1": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "duo-chat-gpt-5-1",
        "knowledge": "2024-09-30",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.1)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-22",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-2": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "duo-chat-gpt-5-2",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-23",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.2)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-2-codex": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "duo-chat-gpt-5-2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.2 Codex)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-22",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-3-codex": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "duo-chat-gpt-5-3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.3 Codex)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-4": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "duo-chat-gpt-5-4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.4)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-4-mini": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-mini",
        "id": "duo-chat-gpt-5-4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.4 Mini)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-4-nano": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-nano",
        "id": "duo-chat-gpt-5-4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.4 Nano)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-5": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "duo-chat-gpt-5-5",
        "knowledge": "2025-08-31",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5.5)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-codex": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "duo-chat-gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5 Codex)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-22",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-gpt-5-mini": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-mini",
        "id": "duo-chat-gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (GPT-5 Mini)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-22",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "duo-chat-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2026-01-08",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Haiku 4.5)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-08",
        "temperature": true,
        "tool_call": true
      },
      "duo-chat-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "duo-chat-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2026-01-08",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Opus 4.5)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-08",
        "temperature": true,
        "tool_call": true
      },
      "duo-chat-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "duo-chat-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Opus 4.6)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "duo-chat-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "duo-chat-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Opus 4.7)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "duo-chat-opus-4-8",
        "knowledge": "2026-01-31",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Opus 4.8)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "duo-chat-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "duo-chat-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2026-01-08",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Sonnet 4.5)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-08",
        "temperature": true,
        "tool_call": true
      },
      "duo-chat-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "duo-chat-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Sonnet 4.6)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "duo-chat-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "duo-chat-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agentic Chat (Claude Sonnet 5)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "GitLab Duo",
    "npm": "gitlab-ai-provider"
  },
  "gmicloud": {
    "api": "https://api.gmi-serving.com/v1",
    "doc": "https://docs.gmicloud.ai/inference-engine/api-reference/llm-api-reference",
    "env": [
      "GMICLOUD_API_KEY"
    ],
    "id": "gmicloud",
    "models": {
      "Qwen/Qwen3.7-Max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.7-Max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 409600,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.45,
          "input": 4.5,
          "output": 22.5
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 409600,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 409600,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.022,
          "input": 0.112,
          "output": 0.224
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-ai/DeepSeek-V4-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048575,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.116,
          "input": 1.392,
          "output": 2.784
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.144,
          "input": 0.855,
          "output": 3.6
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code-highspeed": {
        "attachment": true,
        "cost": {
          "cache_read": 0.38,
          "input": 1.9,
          "output": 8
        },
        "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code Highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "zai-org/GLM-5-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 1.92
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "zai-org/GLM-5-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.182,
          "input": 0.98,
          "output": 3.08
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.182,
          "input": 0.979,
          "output": 3.08
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "GMI Cloud",
    "npm": "@ai-sdk/openai-compatible"
  },
  "google": {
    "doc": "https://ai.google.dev/gemini-api/docs/models",
    "env": [
      "GOOGLE_API_KEY",
      "GOOGLE_GENERATIVE_AI_API_KEY",
      "GEMINI_API_KEY"
    ],
    "id": "google",
    "models": {
      "gemini-2.0-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Earlier Gemini Flash workhorse for responsive multimodal apps and tool use",
        "family": "gemini-flash",
        "id": "gemini-2.0-flash",
        "knowledge": "2024-06",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-11",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.0-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gemini-flash-lite",
        "id": "gemini-2.0-flash-lite",
        "knowledge": "2024-06",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash-Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-11",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-image": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "output": 30
        },
        "description": "Nano Banana image model for fast generation, edits, and character-consistent assets",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash-image",
        "knowledge": "2025-06",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-26",
        "temperature": true,
        "tool_call": false
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "input_audio": 0.3,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-preview-tts": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash-preview-tts",
        "knowledge": "2025-01",
        "last_updated": "2025-05-01",
        "limit": {
          "context": 8192,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Gemini 2.5 Flash Preview TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-01",
        "temperature": true,
        "tool_call": false
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro-preview-tts": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 20
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gemini-flash",
        "id": "gemini-2.5-pro-preview-tts",
        "knowledge": "2025-01",
        "last_updated": "2025-05-01",
        "limit": {
          "context": 8192,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Gemini 2.5 Pro Preview TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-01",
        "temperature": true,
        "tool_call": false
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro-image-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 120
        },
        "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits",
        "family": "gemini-pro",
        "id": "gemini-3-pro-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": false
      },
      "gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "gemini-3-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-image-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 60
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini-flash",
        "id": "gemini-3.1-flash-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": false
      },
      "gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview-customtools",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "input_audio": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-embedding-001": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "gemini",
        "id": "gemini-embedding-001",
        "knowledge": "2025-05",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 2048,
          "output": 1
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Embedding 001",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": false,
        "tool_call": false
      },
      "gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-flash-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-flash-lite-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-flash-lite-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash-Lite Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-26b-a4b-it": {
        "attachment": true,
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-31b-it": {
        "attachment": true,
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Google",
    "npm": "@ai-sdk/google"
  },
  "google-vertex": {
    "doc": "https://cloud.google.com/vertex-ai/generative-ai/docs/models",
    "env": [
      "GOOGLE_VERTEX_PROJECT",
      "GOOGLE_VERTEX_LOCATION",
      "GOOGLE_APPLICATION_CREDENTIALS"
    ],
    "id": "google-vertex",
    "models": {
      "claude-3-5-haiku@20241022": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-3-5-haiku@20241022",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": false,
        "release_date": "2024-10-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5@20251001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5@20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1@20250805": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1@20250805",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5@20251101": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5@20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6@default",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7@default",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8@default",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4@20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4@20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5@20250929": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5@20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6@default",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4@20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4@20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5@default",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google-vertex/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      },
      "deepseek-ai/deepseek-v3.1-maas": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 1.7
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/deepseek-v3.1-maas",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/deepseek-v3.2-maas": {
        "attachment": false,
        "cost": {
          "cache_read": 0.056,
          "input": 0.56,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/deepseek-v3.2-maas",
        "last_updated": "2026-04-04",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "cache_write": 0.383,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "input_audio": 0.3,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-tts": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash-tts",
        "knowledge": "2025-01",
        "last_updated": "2025-12-10",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Gemini 2.5 Flash TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-30",
        "temperature": false,
        "tool_call": false
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro-tts": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 20
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro-tts",
        "knowledge": "2025-01",
        "last_updated": "2025-12-10",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Gemini 2.5 Pro TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-30",
        "temperature": false,
        "tool_call": false
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview-customtools",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "input_audio": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-embedding-001": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "gemini",
        "id": "gemini-embedding-001",
        "knowledge": "2025-05",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 2048,
          "output": 1
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Embedding 001",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": false,
        "tool_call": false
      },
      "gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "cache_write": 0.383,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-flash-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-25",
        "temperature": true,
        "tool_call": true
      },
      "gemini-flash-lite-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-flash-lite-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash-Lite Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.3-70b-instruct-maas": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.3-70b-instruct-maas",
        "knowledge": "2023-12",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": false,
        "release_date": "2025-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-4-maverick-17b-128e-instruct-maas": {
        "attachment": true,
        "cost": {
          "input": 0.35,
          "output": 1.15
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta/llama-4-maverick-17b-128e-instruct-maas",
        "knowledge": "2024-08",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 524288,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": false,
        "release_date": "2025-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking-maas": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking-maas",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b-maas": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b-maas",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b-maas": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.25
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b-maas",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-instruct-2507-maas": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.88
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-instruct-2507-maas",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-maas": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "zai-org/glm-4.7-maas",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-01-06",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-5-maas": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5-maas",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "provider": {
          "api": "https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi",
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Vertex",
    "npm": "@ai-sdk/google-vertex"
  },
  "google-vertex-anthropic": {
    "doc": "https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude",
    "env": [
      "GOOGLE_VERTEX_PROJECT",
      "GOOGLE_VERTEX_LOCATION",
      "GOOGLE_APPLICATION_CREDENTIALS"
    ],
    "id": "google-vertex-anthropic",
    "models": {
      "claude-3-5-haiku@20241022": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-3-5-haiku@20241022",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5@20251001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5@20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1@20250805": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1@20250805",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5@20251101": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5@20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6@default",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7@default",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8@default",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4@20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4@20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5@20250929": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5@20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6@default",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4@20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4@20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5@default": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5@default",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Vertex (Anthropic)",
    "npm": "@ai-sdk/google-vertex/anthropic"
  },
  "groq": {
    "doc": "https://console.groq.com/docs/models",
    "env": [
      "GROQ_API_KEY"
    ],
    "id": "groq",
    "models": {
      "canopylabs/orpheus-arabic-saudi": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "canopylabs",
        "id": "canopylabs/orpheus-arabic-saudi",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 4000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Canopy Labs Orpheus Arabic Saudi",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "status": "beta",
        "temperature": false,
        "tool_call": false
      },
      "canopylabs/orpheus-v1-english": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "canopylabs",
        "id": "canopylabs/orpheus-v1-english",
        "last_updated": "2025-12-19",
        "limit": {
          "context": 4000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Canopy Labs Orpheus V1 English",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-19",
        "status": "beta",
        "temperature": false,
        "tool_call": false
      },
      "groq/compound": {
        "attachment": false,
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "groq",
        "id": "groq/compound",
        "last_updated": "2025-09-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Compound",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-04",
        "temperature": true,
        "tool_call": false
      },
      "groq/compound-mini": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "groq",
        "id": "groq/compound-mini",
        "last_updated": "2025-09-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Compound Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-04",
        "temperature": true,
        "tool_call": false
      },
      "llama-3.1-8b-instant": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.08
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "llama-3.1-8b-instant",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-versatile": {
        "attachment": false,
        "cost": {
          "input": 0.59,
          "output": 0.79
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-versatile",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-4-scout-17b-16e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.11,
          "output": 0.34
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta-llama/llama-4-scout-17b-16e-instruct",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-prompt-guard-2-22m": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.03
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "meta-llama/llama-prompt-guard-2-22m",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 512,
          "output": 512
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama Prompt Guard 2 22M",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-29",
        "status": "beta",
        "temperature": false,
        "tool_call": false
      },
      "meta-llama/llama-prompt-guard-2-86m": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "meta-llama/llama-prompt-guard-2-86m",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 512,
          "output": 512
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Prompt Guard 2 86M",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-29",
        "status": "beta",
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-10-21",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0375,
          "input": 0.075,
          "output": 0.3
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-safeguard-20b": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-safeguard-20b",
        "last_updated": "2026-06-29",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Safety GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-29",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.59
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-32b",
        "last_updated": "2025-06-12",
        "limit": {
          "context": 131072,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "default"
            ]
          }
        ],
        "release_date": "2025-06-11",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      },
      "whisper-large-v3": {
        "attachment": false,
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "whisper-large-v3",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-09-01",
        "temperature": true,
        "tool_call": false
      },
      "whisper-large-v3-turbo": {
        "attachment": false,
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "whisper-large-v3-turbo",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper Large V3 Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Groq",
    "npm": "@ai-sdk/groq"
  },
  "helicone": {
    "api": "https://ai-gateway.helicone.ai/v1",
    "doc": "https://helicone.ai/models",
    "env": [
      "HELICONE_API_KEY"
    ],
    "id": "helicone",
    "models": {
      "chatgpt-4o-latest": {
        "attachment": false,
        "cost": {
          "cache_read": 2.5,
          "input": 5,
          "output": 20
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "chatgpt-4o-latest",
        "knowledge": "2024-08",
        "last_updated": "2024-08-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI ChatGPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-14",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-haiku-20240307": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-3-haiku-20240307",
        "knowledge": "2024-03",
        "last_updated": "2024-03-07",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 3 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-07",
        "temperature": true,
        "tool_call": true
      },
      "claude-3.5-haiku": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.7999999999999999,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-3.5-haiku",
        "knowledge": "2024-10",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 3.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-3.5-sonnet-v2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.30000000000000004,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-3.5-sonnet-v2",
        "knowledge": "2024-10",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 3.5 Sonnet v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-3.7-sonnet": {
        "attachment": false,
        "cost": {
          "cache_read": 0.30000000000000004,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-3.7-sonnet",
        "knowledge": "2025-02",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 3.7 Sonnet",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": true
      },
      "claude-4.5-haiku": {
        "attachment": false,
        "cost": {
          "cache_read": 0.09999999999999999,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-4.5-haiku",
        "knowledge": "2025-10",
        "last_updated": "2025-10-01",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 4.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-4.5-opus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-4.5-opus",
        "knowledge": "2025-11",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-4.5-sonnet": {
        "attachment": false,
        "cost": {
          "cache_read": 0.30000000000000004,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-4.5-sonnet",
        "knowledge": "2025-09",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": false,
        "cost": {
          "cache_read": 0.09999999999999999,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-10",
        "last_updated": "2025-10-01",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 4.5 Haiku (20251001)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4": {
        "attachment": false,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4",
        "knowledge": "2025-05",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-14",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1": {
        "attachment": false,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1",
        "knowledge": "2025-08",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": false,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1-20250805",
        "knowledge": "2025-08",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.1 (20250805)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4": {
        "attachment": false,
        "cost": {
          "cache_read": 0.30000000000000004,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4",
        "knowledge": "2025-05",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-14",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": false,
        "cost": {
          "cache_read": 0.30000000000000004,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "knowledge": "2025-09",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet 4.5 (20250929)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.13
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1-distill-llama-70b",
        "knowledge": "2025-01",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Llama 70B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-reasoner": {
        "attachment": false,
        "cost": {
          "cache_read": 0.07,
          "input": 0.56,
          "output": 1.68
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-reasoner",
        "knowledge": "2025-01",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Reasoner",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-tng-r1t2-chimera": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek-thinking",
        "id": "deepseek-tng-r1t2-chimera",
        "knowledge": "2025-07",
        "last_updated": "2025-07-02",
        "limit": {
          "context": 130000,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek TNG R1T2 Chimera",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-02",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.07,
          "input": 0.56,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3",
        "knowledge": "2024-12",
        "last_updated": "2024-12-26",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-26",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.1-terminus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.21600000000000003,
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.1-terminus",
        "knowledge": "2025-09",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Terminus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-22",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.41
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2025-09",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-22",
        "temperature": true,
        "tool_call": true
      },
      "ernie-4.5-21b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "ernie",
        "id": "ernie-4.5-21b-a3b-thinking",
        "knowledge": "2025-03",
        "last_updated": "2025-03-16",
        "limit": {
          "context": 128000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu Ernie 4.5 21B A3B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-16",
        "temperature": true,
        "tool_call": false
      },
      "gemini-2.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "cache_write": 0.3,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-06",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite": {
        "attachment": false,
        "cost": {
          "cache_read": 0.024999999999999998,
          "cache_write": 0.09999999999999999,
          "input": 0.09999999999999999,
          "output": 0.39999999999999997
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite",
        "knowledge": "2025-07",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemini 2.5 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-22",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3125,
          "cache_write": 1.25,
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-06",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.19999999999999998,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3-pro-preview",
        "knowledge": "2025-11",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "temperature": true,
        "tool_call": true
      },
      "gemma-3-12b-it": {
        "attachment": false,
        "cost": {
          "input": 0.049999999999999996,
          "output": 0.09999999999999999
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-3-12b-it",
        "knowledge": "2024-12",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 3 12B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": false
      },
      "gemma2-9b-it": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0.03
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma2-9b-it",
        "knowledge": "2024-06",
        "last_updated": "2024-06-25",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-25",
        "temperature": true,
        "tool_call": false
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "input": 0.44999999999999996,
          "output": 1.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2024-07",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Zai GLM-4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2025-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.09999999999999999,
          "input": 0.39999999999999997,
          "output": 1.5999999999999999
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2025-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-4.1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini-2025-04-14": {
        "attachment": false,
        "cost": {
          "cache_read": 0.09999999999999999,
          "input": 0.39999999999999997,
          "output": 1.5999999999999999
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini-2025-04-14",
        "knowledge": "2025-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-4.1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": false,
        "cost": {
          "cache_read": 0.024999999999999998,
          "input": 0.09999999999999999,
          "output": 0.39999999999999997
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-4.1-nano",
        "knowledge": "2025-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-4.1 Nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": false,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2024-05",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-4o-mini",
        "knowledge": "2024-07",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-4o-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12500000000000003,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-chat-latest": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12500000000000003,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5-chat-latest",
        "knowledge": "2024-09",
        "last_updated": "2024-09-30",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-5 Chat Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-30",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12500000000000003,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5-codex",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Codex",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.024999999999999998,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-5 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": false,
        "cost": {
          "cache_read": 0.005,
          "input": 0.049999999999999996,
          "output": 0.39999999999999997
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-5 Nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": false,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": false
      },
      "gpt-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12500000000000003,
          "input": 1.25,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "OpenAI GPT-5.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-chat-latest": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12500000000000003,
          "input": 1.25,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-codex",
        "id": "gpt-5.1-chat-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "OpenAI GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12500000000000003,
          "input": 1.25,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "OpenAI: GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.024999999999999998,
          "input": 0.25,
          "output": 2
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "OpenAI: GPT-5.1 Codex Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.16
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-OSS 120b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-06-01",
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.049999999999999996,
          "output": 0.19999999999999998
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-20b",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT-OSS 20b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-06-01",
        "temperature": true,
        "tool_call": true
      },
      "grok-3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.75,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-3",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI Grok 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "temperature": true,
        "tool_call": true
      },
      "grok-3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-3-mini",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI Grok 3 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "temperature": true,
        "tool_call": true
      },
      "grok-4": {
        "attachment": false,
        "cost": {
          "cache_read": 0.75,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4",
        "knowledge": "2024-07",
        "last_updated": "2024-07-09",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI Grok 4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-09",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.049999999999999996,
          "input": 0.19999999999999998,
          "output": 0.5
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "grok-4-1-fast-non-reasoning",
        "knowledge": "2025-11",
        "last_updated": "2025-11-17",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "xAI Grok 4.1 Fast Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-17",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-reasoning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.049999999999999996,
          "input": 0.19999999999999998,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-reasoning",
        "knowledge": "2025-11",
        "last_updated": "2025-11-17",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI Grok 4.1 Fast Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-17",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-non-reasoning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.049999999999999996,
          "input": 0.19999999999999998,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-non-reasoning",
        "knowledge": "2025-09",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI Grok 4 Fast Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-reasoning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.049999999999999996,
          "input": 0.19999999999999998,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-reasoning",
        "knowledge": "2025-09",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI: Grok 4 Fast Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": true
      },
      "grok-code-fast-1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.19999999999999998,
          "output": 1.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-code-fast-1",
        "knowledge": "2024-08",
        "last_updated": "2024-08-25",
        "limit": {
          "context": 256000,
          "output": 10000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI Grok Code Fast 1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-25",
        "temperature": true,
        "tool_call": true
      },
      "hermes-2-pro-llama-3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.14
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "hermes-2-pro-llama-3-8b",
        "knowledge": "2024-05",
        "last_updated": "2024-05-27",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 2 Pro Llama 3 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-27",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-0711": {
        "attachment": false,
        "cost": {
          "input": 0.5700000000000001,
          "output": 2.3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0711",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 (07/11)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "cache_read": 0.39999999999999997,
          "input": 0.5,
          "output": 2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0905",
        "knowledge": "2025-09",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 (09/05)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.48,
          "output": 2
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "knowledge": "2025-11",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.1-8b-instant": {
        "attachment": false,
        "cost": {
          "input": 0.049999999999999996,
          "output": 0.08
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "llama-3.1-8b-instant",
        "knowledge": "2024-07",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 131072,
          "output": 32678
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3.1 8B Instant",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.049999999999999996
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.1-8b-instruct",
        "knowledge": "2024-07",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3.1 8B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.1-8b-instruct-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.03
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "llama-3.1-8b-instruct-turbo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3.1 8B Instruct Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.39
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2024-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 16400
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3.3 70B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-versatile": {
        "attachment": false,
        "cost": {
          "input": 0.59,
          "output": 0.7899999999999999
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-versatile",
        "knowledge": "2024-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 32678
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3.3 70B Versatile",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-maverick": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "llama-4-maverick",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 4 Maverick 17B 128E",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-scout": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.3
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "llama-4-scout",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 4 Scout 17B 16E",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "llama-guard-4": {
        "attachment": false,
        "cost": {
          "input": 0.21,
          "output": 0.21
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "llama-guard-4",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 131072,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama Guard 4 12B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": false
      },
      "llama-prompt-guard-2-22m": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0.01
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "llama-prompt-guard-2-22m",
        "knowledge": "2024-10",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 512,
          "output": 2
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama Prompt Guard 2 22M",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": false
      },
      "llama-prompt-guard-2-86m": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0.01
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "llama-prompt-guard-2-86m",
        "knowledge": "2024-10",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 512,
          "output": 2
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama Prompt Guard 2 86M",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": false
      },
      "mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-2411",
        "knowledge": "2024-07",
        "last_updated": "2024-07-24",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral-Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-24",
        "temperature": true,
        "tool_call": true
      },
      "mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 20,
          "output": 40
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistral-nemo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16400
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": false
      },
      "mistral-small": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.2
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small",
        "knowledge": "2025-03",
        "last_updated": "2025-06-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "temperature": true,
        "tool_call": true
      },
      "o1": {
        "attachment": false,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o1",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": false
      },
      "o1-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o1-mini",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o1-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": false
      },
      "o3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o3",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "temperature": false,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o3-mini",
        "knowledge": "2023-10",
        "last_updated": "2023-10-01",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-10-01",
        "temperature": false,
        "tool_call": true
      },
      "o3-pro": {
        "attachment": false,
        "cost": {
          "input": 20,
          "output": 80
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "o3-pro",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "temperature": false,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "o4-mini",
        "knowledge": "2024-06",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o4 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "temperature": false,
        "tool_call": true
      },
      "qwen2.5-coder-7b-fast": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.09
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen2.5-coder-7b-fast",
        "knowledge": "2024-09",
        "last_updated": "2024-09-15",
        "limit": {
          "context": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 Coder 7B fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-15",
        "temperature": true,
        "tool_call": false
      },
      "qwen3-235b-a22b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2.9000000000000004
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-235b-a22b-thinking",
        "knowledge": "2025-07",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 262144,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": false
      },
      "qwen3-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.29
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-30b-a3b",
        "knowledge": "2025-06",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 41000,
          "output": 41000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.59
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-32b",
        "knowledge": "2025-04",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 131072,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.95
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder",
        "knowledge": "2025-07",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.09999999999999999,
          "output": 0.3
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-07",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B A3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 1.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-instruct",
        "knowledge": "2025-01",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 262000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-235b-a22b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-235b-a22b-instruct",
        "knowledge": "2025-09",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "sonar": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar",
        "id": "sonar",
        "knowledge": "2025-01",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 127000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-27",
        "temperature": true,
        "tool_call": false
      },
      "sonar-deep-research": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar-deep-research",
        "id": "sonar-deep-research",
        "knowledge": "2025-01",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 127000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Sonar Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-01-27",
        "temperature": true,
        "tool_call": false
      },
      "sonar-pro": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "family": "sonar-pro",
        "id": "sonar-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-27",
        "temperature": true,
        "tool_call": false
      },
      "sonar-reasoning": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Web-grounded reasoning model for multi-step research and cited answers",
        "family": "sonar-reasoning",
        "id": "sonar-reasoning",
        "knowledge": "2025-01",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 127000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Sonar Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-27",
        "temperature": true,
        "tool_call": false
      },
      "sonar-reasoning-pro": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Web-grounded reasoning model for multi-step research and cited answers",
        "family": "sonar-reasoning",
        "id": "sonar-reasoning-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 127000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-01-27",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Helicone",
    "npm": "@ai-sdk/openai-compatible"
  },
  "hpc-ai": {
    "api": "https://api.hpc-ai.com/inference/v1",
    "doc": "https://www.hpc-ai.com/doc/docs/quickstart/",
    "env": [
      "HPC_AI_API_KEY"
    ],
    "id": "hpc-ai",
    "models": {
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax-m2.5",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.3,
          "output": 1.5
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "zai-org/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.133,
          "input": 0.615,
          "output": 2.46
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-01",
        "limit": {
          "context": 202000,
          "output": 202000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "HPC-AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "huggingface": {
    "api": "https://router.huggingface.co/v1",
    "doc": "https://huggingface.co/docs/inference-providers",
    "env": [
      "HF_TOKEN"
    ],
    "id": "huggingface",
    "models": {
      "MiniMaxAI/MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-10",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M3": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 524288,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 40960,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 3
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B-Thinking-2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-32B": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.59
        },
        "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-32B",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-30B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.26
        },
        "description": "Smaller Qwen coder for efficient local agents and repo-level fixes",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 66536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-480B-A35B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-Next": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-Next",
        "knowledge": "2025-04",
        "last_updated": "2026-02-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-Next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-03",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Embedding-4B": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "Qwen/Qwen3-Embedding-4B",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32000,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Embedding 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": false
      },
      "Qwen/Qwen3-Embedding-8B": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "Qwen/Qwen3-Embedding-8B",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Embedding 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": false
      },
      "Qwen/Qwen3-Next-80B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 262144,
          "output": 66536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next-80B-A3B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Next-80B-A3B-Thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-Next-80B-A3B-Thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next-80B-A3B-Thinking",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-122B-A10B": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-122B-A10B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-27B": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-27B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-35B-A3B": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-35B-A3B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-9B": {
        "attachment": true,
        "cost": {
          "input": 0.17,
          "output": 0.25
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-9B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-27B": {
        "attachment": true,
        "cost": {
          "input": 0.47,
          "output": 3.19
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-27B",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.95
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-35B-A3B",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "XiaomiMiMo/MiMo-V2-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "XiaomiMiMo/MiMo-V2-Flash",
        "knowledge": "2024-12",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 262144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "XiaomiMiMo/MiMo-V2.5-Pro": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "XiaomiMiMo/MiMo-V2.5-Pro",
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.5
        },
        "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 64000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1-0528": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 5
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1-0528",
        "knowledge": "2025-05",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1-0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 0.4
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-ai/DeepSeek-V4-Flash",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 393216
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26B-A4B-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26B-A4B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.4
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.59,
          "output": 0.79
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2-Instruct": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2-Instruct",
        "knowledge": "2024-10",
        "last_updated": "2025-07-14",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-14",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2-Instruct-0905": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2-Instruct-0905",
        "knowledge": "2024-10",
        "last_updated": "2025-09-04",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2-Instruct-0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-04",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2-Thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/Kimi-K2-Thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2-Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-01",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.7-Code": {
        "attachment": true,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.7-Code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.69
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/Step-3.5-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "id": "stepfun-ai/Step-3.5-Flash",
        "knowledge": "2025-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 262144,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/Step-3.7-Flash": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.15
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "id": "stepfun-ai/Step-3.7-Flash",
        "knowledge": "2026-01-01",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 262144,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "zai-org/GLM-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-Air": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.85
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "zai-org/GLM-4.5-Air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5V": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "zai-org/GLM-4.5V",
        "knowledge": "2025-04",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.6": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.2
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "zai-org/GLM-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7-Flash": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "zai-org/GLM-4.7-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-08",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-03",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-03",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Hugging Face",
    "npm": "@ai-sdk/openai-compatible"
  },
  "iflowcn": {
    "api": "https://apis.iflow.cn/v1",
    "doc": "https://platform.iflow.cn/en/docs",
    "env": [
      "IFLOW_API_KEY"
    ],
    "id": "iflowcn",
    "models": {
      "deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-r1",
        "knowledge": "2024-12",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3",
        "knowledge": "2024-10",
        "last_updated": "2024-12-26",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-26",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2-Exp",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2024-10",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0905",
        "knowledge": "2024-12",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2-0905",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen3-235b-a22b-thinking-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B-Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-32b",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3-max",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max-preview": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3-max-preview",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Max-Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-plus": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-plus",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL-Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "iFlow",
    "npm": "@ai-sdk/openai-compatible"
  },
  "inception": {
    "api": "https://api.inceptionlabs.ai/v1/",
    "doc": "https://platform.inceptionlabs.ai/docs",
    "env": [
      "INCEPTION_API_KEY"
    ],
    "id": "inception",
    "models": {
      "mercury-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "mercury",
        "id": "mercury-2",
        "knowledge": "2025-01-01",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 128000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mercury-edit-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "mercury-edit-2",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury Edit 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-30",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Inception",
    "npm": "@ai-sdk/openai-compatible"
  },
  "inceptron": {
    "api": "https://api.inceptron.io/v1",
    "doc": "https://docs.inceptron.io",
    "env": [
      "INCEPTRON_API_KEY"
    ],
    "id": "inceptron",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0,
          "input": 0.15,
          "output": 0.9
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 0.66,
          "output": 3.5
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6-Fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4,
          "cache_write": 0,
          "input": 1.32,
          "output": 7
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6-Fast",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 Fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "status": "alpha",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.7-Code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 0.75,
          "output": 3.5
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.7-Code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "zai-org/GLM-5.1-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 202752
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.2,
          "output": 4.2
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Inceptron",
    "npm": "@ai-sdk/openai-compatible"
  },
  "inference": {
    "api": "https://inference.net/v1",
    "doc": "https://inference.net/models",
    "env": [
      "INFERENCE_API_KEY"
    ],
    "id": "inference",
    "models": {
      "google/gemma-3": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.3
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 125000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.025,
          "output": 0.025
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.1-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.055,
          "output": 0.055
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta/llama-3.2-11b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11B Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-1b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0.01
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.2-1b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 1B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.02
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.2-3b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-nemo-12b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.038,
          "output": 0.1
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistral/mistral-nemo-12b-instruct",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo 12B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "osmosis/osmosis-structure-0.6b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "osmosis",
        "id": "osmosis/osmosis-structure-0.6b",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 4000,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Osmosis Structure 0.6B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-7b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen-2.5-7b-vision-instruct",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 125000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 7B Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-embedding-4b": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "qwen/qwen3-embedding-4b",
        "knowledge": "2024-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 32000,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Embedding 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "Inference",
    "npm": "@ai-sdk/openai-compatible"
  },
  "io-net": {
    "api": "https://api.intelligence.io.solutions/api/v1",
    "doc": "https://io.net/docs/guides/intelligence/io-intelligence",
    "env": [
      "IOINTELLIGENCE_API_KEY"
    ],
    "id": "io-net",
    "models": {
      "Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0.44,
          "input": 0.22,
          "output": 0.95
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar",
        "knowledge": "2024-12",
        "last_updated": "2025-01-15",
        "limit": {
          "context": 106000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Coder 480B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-15",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-VL-32B-Instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.1,
          "input": 0.05,
          "output": 0.22
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-VL-32B-Instruct",
        "knowledge": "2024-09",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 VL 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "cache_read": 0.055,
          "cache_write": 0.22,
          "input": 0.11,
          "output": 0.6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "knowledge": "2024-12",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 262144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Next-80B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.2,
          "input": 0.1,
          "output": 0.8
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
        "knowledge": "2024-12",
        "last_updated": "2025-01-10",
        "limit": {
          "context": 262144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Next 80B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-10",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1-0528": {
        "attachment": false,
        "cost": {
          "cache_read": 1,
          "cache_write": 4,
          "input": 2,
          "output": 8.75
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1-0528",
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.2-90B-Vision-Instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "cache_write": 0.7,
          "input": 0.35,
          "output": 0.4
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta-llama/Llama-3.2-90B-Vision-Instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 90B Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.065,
          "cache_write": 0.26,
          "input": 0.13,
          "output": 0.38
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.075,
          "cache_write": 0.3,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
        "knowledge": "2024-12",
        "last_updated": "2025-01-15",
        "limit": {
          "context": 430000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B 128E Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-15",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Devstral-Small-2505": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.1,
          "input": 0.05,
          "output": 0.22
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistralai/Devstral-Small-2505",
        "knowledge": "2024-12",
        "last_updated": "2025-05-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small 2505",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Magistral-Small-2506": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 1,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral-small",
        "id": "mistralai/Magistral-Small-2506",
        "knowledge": "2025-01",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Small 2506",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Mistral-Large-Instruct-2411": {
        "attachment": false,
        "cost": {
          "cache_read": 1,
          "cache_write": 4,
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistralai/Mistral-Large-Instruct-2411",
        "knowledge": "2024-10",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large Instruct 2411",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Mistral-Nemo-Instruct-2407": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.04,
          "input": 0.02,
          "output": 0.04
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistralai/Mistral-Nemo-Instruct-2407",
        "knowledge": "2024-05",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo Instruct 2407",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2-Instruct-0905": {
        "attachment": false,
        "cost": {
          "cache_read": 0.195,
          "cache_write": 0.78,
          "input": 0.39,
          "output": 1.9
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2-Instruct-0905",
        "knowledge": "2024-08",
        "last_updated": "2024-09-05",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-05",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2-Thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.275,
          "cache_write": 1.1,
          "input": 0.55,
          "output": 2.25
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/Kimi-K2-Thinking",
        "knowledge": "2024-08",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.08,
          "input": 0.04,
          "output": 0.4
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 131072,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 120B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "cache_write": 0.06,
          "input": 0.03,
          "output": 0.14
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 64000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 20B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.8,
          "input": 0.4,
          "output": 1.75
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.6",
        "knowledge": "2024-10",
        "last_updated": "2024-11-15",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-15",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "IO.NET",
    "npm": "@ai-sdk/openai-compatible"
  },
  "jiekou": {
    "api": "https://api.jiekou.ai/openai",
    "doc": "https://docs.jiekou.ai/docs/support/quickstart?utm_source=github_models.dev",
    "env": [
      "JIEKOU_API_KEY"
    ],
    "id": "jiekou",
    "models": {
      "baidu/ernie-4.5-300b-a47b-paddle": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.1
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "ernie",
        "id": "baidu/ernie-4.5-300b-a47b-paddle",
        "last_updated": "2026-01",
        "limit": {
          "context": 123000,
          "output": 12000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 300B A47B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-vl-424b-a47b": {
        "attachment": true,
        "cost": {
          "input": 0.42,
          "output": 1.25
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "ernie",
        "id": "baidu/ernie-4.5-vl-424b-a47b",
        "last_updated": "2026-01",
        "limit": {
          "context": 123000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 VL 424B A47B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "input": 0.9,
          "output": 4.5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "last_updated": "2026-01",
        "limit": {
          "context": 20000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-haiku-4-5-20251001",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "input": 13.5,
          "output": 67.5
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1-20250805",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-1-20250805",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 13.5,
          "output": 67.5
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-20250514",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-20250514",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "input": 4.5,
          "output": 22.5
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5-20251101",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-5-20251101",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-opus-4-6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 2.7,
          "output": 13.5
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-20250514",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-20250514",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "input": 2.7,
          "output": 13.5
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-sonnet-4-5-20250929",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.5
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-0528",
        "last_updated": "2026-01",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.14
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3-0324",
        "last_updated": "2026-01",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1",
        "last_updated": "2026-01",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32767,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.27,
          "output": 2.25
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite-preview-06-17": {
        "attachment": true,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite-preview-06-17",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "video",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-lite-preview-06-17",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite-preview-09-2025": {
        "attachment": true,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite-preview-09-2025",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-lite-preview-09-2025",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-preview-05-20": {
        "attachment": true,
        "cost": {
          "input": 0.135,
          "output": 3.15
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash-preview-05-20",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-preview-05-20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro-preview-06-05": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro-preview-06-05",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-pro-preview-06-05",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-3-flash-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 1.8,
          "output": 10.8
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3-pro-preview",
        "last_updated": "2026-01",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-3-pro-preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-chat-latest": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5-chat-latest",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-chat-latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-codex": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5-codex",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-codex",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "input": 0.225,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "input": 0.045,
          "output": 0.36
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 13.5,
          "output": 108
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "last_updated": "2026-02",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.1-codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "input": 1.125,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-max",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.1-codex-max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "input": 0.225,
          "output": 1.8
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.1-codex-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "input": 1.575,
          "output": 12.6
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.2-codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 18.9,
          "output": 151.2
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5.2-pro",
        "last_updated": "2026-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.2-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-0709": {
        "attachment": true,
        "cost": {
          "input": 2.7,
          "output": 13.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-0709",
        "last_updated": "2026-01",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-0709",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.45
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-non-reasoning",
        "last_updated": "2026-01",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-1-fast-non-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.45
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-reasoning",
        "last_updated": "2026-01",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-1-fast-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.45
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-non-reasoning",
        "last_updated": "2026-01",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-fast-non-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-fast-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.45
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-fast-reasoning",
        "last_updated": "2026-01",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-4-fast-reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-code-fast-1": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 1.35
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-code-fast-1",
        "last_updated": "2026-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "grok-code-fast-1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.1",
        "last_updated": "2026-01",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 131071,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimaxai/minimax-m1-80k": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimaxai/minimax-m1-80k",
        "last_updated": "2026-01",
        "limit": {
          "context": 1000000,
          "output": 40000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-0905",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.57,
          "output": 2.3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-instruct",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 262143,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "o3": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 40
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "o3",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o3-mini",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o4-mini",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-fp8",
        "last_updated": "2026-01",
        "limit": {
          "context": 40960,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.8
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-instruct-2507",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 3
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-thinking-2507",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22b Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.45
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-30b-a3b-fp8",
        "last_updated": "2026-01",
        "limit": {
          "context": 40960,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-32b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.45
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-32b-fp8",
        "last_updated": "2026-01",
        "limit": {
          "context": 40960,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 1.2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-480b-a35b-instruct",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-next",
        "last_updated": "2026-02",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen/qwen3-coder-next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-next-80b-a3b-instruct",
        "last_updated": "2026-01",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-next-80b-a3b-thinking",
        "last_updated": "2026-01",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xiaomimimo/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomimimo/mimo-v2-flash",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "XiaomiMiMo/MiMo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.5",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.5v": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glmv",
        "id": "zai-org/glm-4.5v",
        "last_updated": "2026-01",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.7",
        "last_updated": "2026-01",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "zai-org/glm-4.7-flash",
        "last_updated": "2026-01",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Jiekou.AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "kenari": {
    "api": "https://kenari.id/v1",
    "doc": "https://kenari.id/docs",
    "env": [
      "KENARI_API_KEY"
    ],
    "id": "kenari",
    "models": {
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash:free",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash (Free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro:free",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro (Free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5-1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5-1",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5-2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5-2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-4-mini": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "gpt-5-4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-5": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5-5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-image-2": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "gpt-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 272000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-Image-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open-weight GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "tool_call": true
      },
      "grok-4-3": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "grok-4-3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-build-0-1": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "grok-build-0-1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-6": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2-6",
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-7-code": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2-7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "mimo-v2-5": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "mimo-v2-5",
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-5-pro": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2-5-pro",
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-7-plus": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3-7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Kenari",
    "npm": "@ai-sdk/openai-compatible"
  },
  "kilo": {
    "api": "https://api.kilo.ai/api/gateway",
    "doc": "https://kilo.ai",
    "env": [
      "KILO_API_KEY"
    ],
    "id": "kilo",
    "models": {
      "ai21/jamba-large-1.7": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "ai21/jamba-large-1.7",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AI21: Jamba Large 1.7",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-09",
        "temperature": true,
        "tool_call": true
      },
      "aion-labs/aion-1.0": {
        "attachment": false,
        "cost": {
          "input": 4,
          "output": 8
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "aion-labs/aion-1.0",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AionLabs: Aion-1.0",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-05",
        "temperature": true,
        "tool_call": false
      },
      "aion-labs/aion-1.0-mini": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 1.4
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "aion-labs/aion-1.0-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AionLabs: Aion-1.0-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-05",
        "temperature": true,
        "tool_call": false
      },
      "aion-labs/aion-2.0": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.6
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "aion-labs/aion-2.0",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AionLabs: Aion-2.0",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-24",
        "temperature": true,
        "tool_call": false
      },
      "aion-labs/aion-rp-llama-3.1-8b": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.6
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "aion-labs/aion-rp-llama-3.1-8b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AionLabs: Aion-RP 1.0 (8B)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-05",
        "temperature": true,
        "tool_call": false
      },
      "alfredpros/codellama-7b-instruct-solidity": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.2
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "alfredpros/codellama-7b-instruct-solidity",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AlfredPros: CodeLLaMa 7B Instruct Solidity",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": false
      },
      "allenai/olmo-3-32b-think": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "allenai/olmo-3-32b-think",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AllenAI: Olmo 3 32B Think",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-22",
        "temperature": true,
        "tool_call": false
      },
      "amazon/nova-2-lite-v1": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "amazon/nova-2-lite-v1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon: Nova 2 Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-lite-v1": {
        "attachment": true,
        "cost": {
          "input": 0.06,
          "output": 0.24
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "amazon/nova-lite-v1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 300000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon: Nova Lite 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-micro-v1": {
        "attachment": false,
        "cost": {
          "input": 0.035,
          "output": 0.14
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "amazon/nova-micro-v1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon: Nova Micro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-premier-v1": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 12.5
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "amazon/nova-premier-v1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon: Nova Premier 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-pro-v1": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 3.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "amazon/nova-pro-v1",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon: Nova Pro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "anthracite-org/magnum-v4-72b": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "anthracite-org/magnum-v4-72b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 16384,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magnum v4 72B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": false
      },
      "anthropic/claude-3-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "anthropic/claude-3-haiku",
        "last_updated": "2024-03-07",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 3 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-07",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3.5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "anthropic/claude-3.5-haiku",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude 3.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "anthropic/claude-haiku-4.5",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 3,
          "cache_write": 37.5,
          "input": 30,
          "output": 150
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.6-fast",
        "knowledge": "2025-05-31",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.6 (Fast)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-07",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.7",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 3,
          "cache_write": 37.5,
          "input": 30,
          "output": 150
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.7-fast",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus 4.7 (Fast)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-12",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4.5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/coder-large": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 0.8
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "arcee-ai/coder-large",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Arcee AI: Coder Large",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-06",
        "temperature": true,
        "tool_call": false
      },
      "arcee-ai/maestro-reasoning": {
        "attachment": false,
        "cost": {
          "input": 0.9,
          "output": 3.3
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "arcee-ai/maestro-reasoning",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Arcee AI: Maestro Reasoning",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-06",
        "temperature": true,
        "tool_call": false
      },
      "arcee-ai/spotlight": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.18
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "arcee-ai/spotlight",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 65537
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Arcee AI: Spotlight",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-06",
        "temperature": true,
        "tool_call": false
      },
      "arcee-ai/trinity-large-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.85
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "arcee-ai/trinity-large-thinking",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Arcee AI: Trinity Large Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/trinity-mini": {
        "attachment": false,
        "cost": {
          "input": 0.045,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "arcee-ai/trinity-mini",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Arcee AI: Trinity Mini",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12",
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/virtuoso-large": {
        "attachment": false,
        "cost": {
          "input": 0.75,
          "output": 1.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "arcee-ai/virtuoso-large",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Arcee AI: Virtuoso Large",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-06",
        "temperature": true,
        "tool_call": true
      },
      "baidu/cobuddy:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "id": "baidu/cobuddy:free",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: CoBuddy (free)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-06",
        "temperature": false,
        "tool_call": true
      },
      "baidu/ernie-4.5-21b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "baidu/ernie-4.5-21b-a3b",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 120000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: ERNIE 4.5 21B A3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-21b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "baidu/ernie-4.5-21b-a3b-thinking",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: ERNIE 4.5 21B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": false
      },
      "baidu/ernie-4.5-300b-a47b": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.1
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "baidu/ernie-4.5-300b-a47b",
        "last_updated": "2026-01",
        "limit": {
          "context": 123000,
          "output": 12000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: ERNIE 4.5 300B A47B ",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": false
      },
      "baidu/ernie-4.5-vl-28b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.56
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "baidu/ernie-4.5-vl-28b-a3b",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 30000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: ERNIE 4.5 VL 28B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-vl-424b-a47b": {
        "attachment": true,
        "cost": {
          "input": 0.42,
          "output": 1.25
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "baidu/ernie-4.5-vl-424b-a47b",
        "last_updated": "2026-01",
        "limit": {
          "context": 123000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: ERNIE 4.5 VL 424B A47B ",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": false
      },
      "baidu/qianfan-ocr-fast": {
        "attachment": true,
        "cost": {
          "input": 0.68,
          "output": 2.81
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "id": "baidu/qianfan-ocr-fast",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 65536,
          "output": 28672
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baidu: Qianfan-OCR-Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": false
      },
      "bytedance-seed/seed-1.6": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "bytedance-seed/seed-1.6",
        "last_updated": "2025-09",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance Seed: Seed 1.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "bytedance-seed/seed-1.6-flash": {
        "attachment": true,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "bytedance-seed/seed-1.6-flash",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance Seed: Seed 1.6 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "bytedance-seed/seed-2.0-lite": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "bytedance-seed/seed-2.0-lite",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance Seed: Seed-2.0-Lite",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": true
      },
      "bytedance-seed/seed-2.0-mini": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "bytedance-seed/seed-2.0-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance Seed: Seed-2.0-Mini",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-27",
        "temperature": true,
        "tool_call": true
      },
      "bytedance/ui-tars-1.5-7b": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.2
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "bytedance/ui-tars-1.5-7b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 2048
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance: UI-TARS 7B ",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": false
      },
      "cohere/command-a": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "id": "cohere/command-a",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere: Command A",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": false
      },
      "cohere/command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "id": "cohere/command-r-08-2024",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere: Command R (08-2024)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "id": "cohere/command-r-plus-08-2024",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere: Command R+ (08-2024)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r7b-12-2024": {
        "attachment": false,
        "cost": {
          "input": 0.0375,
          "output": 0.15
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "id": "cohere/command-r7b-12-2024",
        "last_updated": "2024-12-02",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere: Command R7B (12-2024)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-02",
        "temperature": true,
        "tool_call": true
      },
      "deepcogito/cogito-v2.1-671b": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 1.25
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "deepcogito/cogito-v2.1-671b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deep Cogito: Cogito v2.1 671B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-chat": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.32,
          "output": 0.89
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-chat",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat-v3-0324": {
        "attachment": false,
        "cost": {
          "cache_read": 0.095,
          "input": 0.2,
          "output": 0.77
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-chat-v3-0324",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3 0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat-v3.1": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.75
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-chat-v3.1",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 32768,
          "output": 7168
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-21",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.5
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek/deepseek-r1",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 64000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.45,
          "output": 2.15
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek/deepseek-r1-0528",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: R1 0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.7,
          "output": 0.8
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek/deepseek-r1-distill-llama-70b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: R1 Distill Llama 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-01-23",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1-distill-qwen-32b": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 0.29
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "deepseek/deepseek-r1-distill-qwen-32b",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: R1 Distill Qwen 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v3.1-terminus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 0.21,
          "output": 0.79
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.1-terminus",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3.1 Terminus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-22",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 0.26,
          "output": 0.38
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-exp": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.41
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-exp",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3.2 Exp",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-speciale": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.4,
          "output": 1.2
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-speciale",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V3.2 Speciale",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "id": "deepseek/deepseek-v4-flash",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V4 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "deepseek/deepseek-v4-pro",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek: DeepSeek V4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "temperature": true,
        "tool_call": true
      },
      "essentialai/rnj-1-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "essentialai/rnj-1-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 6554
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "EssentialAI: Rnj 1 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-05",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.0-flash-001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.083333,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-2.0-flash-001",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.0 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.0-flash-lite-001": {
        "attachment": true,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-2.0-flash-lite-001",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.0 Flash Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.083333,
          "input": 0.3,
          "output": 2.5,
          "reasoning": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-2.5-flash",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-image": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "google/gemini-2.5-flash-image",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Google: Nano Banana (Gemini 2.5 Flash Image)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-08",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.083333,
          "input": 0.1,
          "output": 0.4,
          "reasoning": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-2.5-flash-lite",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.5 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite-preview-09-2025": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.083333,
          "input": 0.1,
          "output": 0.4,
          "reasoning": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-2.5-flash-lite-preview-09-2025",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.5 Flash Lite Preview 09-2025",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-25",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 0.375,
          "input": 1.25,
          "output": 10,
          "reasoning": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-2.5-pro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 0.375,
          "input": 1.25,
          "output": 10,
          "reasoning": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-2.5-pro-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.5 Pro Preview 06-05",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-05",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro-preview-05-06": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 0.375,
          "input": 1.25,
          "output": 10,
          "reasoning": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-2.5-pro-preview-05-06",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 2.5 Pro Preview 05-06",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-06",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.083333,
          "input": 0.5,
          "output": 3,
          "reasoning": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3-flash-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-image-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12,
          "reasoning": 12
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "google/gemini-3-pro-image-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 65536,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-image-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "google/gemini-3.1-flash-image-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Google: Nano Banana 2 (Gemini 3.1 Flash Image Preview)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.08333,
          "input": 0.25,
          "output": 1.5,
          "reasoning": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-3.1-flash-lite",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.5,
          "reasoning": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-3.1-flash-lite-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12,
          "reasoning": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12,
          "reasoning": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview-customtools",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "cache_write": 0.08333,
          "input": 1.5,
          "output": 9,
          "reasoning": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3.5-flash",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-2-27b-it": {
        "attachment": false,
        "cost": {
          "input": 0.65,
          "output": 0.65
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-2-27b-it",
        "last_updated": "2024-06-24",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 2 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-24",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-3-12b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.04,
          "output": 0.13
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3-12b-it",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 3 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.03,
          "output": 0.11
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3-27b-it",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 3 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-12",
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3-4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.04,
          "output": 0.08
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3-4b-it",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 19200
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 3 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-3n-e4b-it": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.04
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3n-e4b-it",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 32768,
          "output": 6554
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 3n 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 4 26B A4B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-03",
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemma 4 31B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "google/lyria-3-clip-preview": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "google/lyria-3-clip-preview",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "audio",
            "text"
          ]
        },
        "name": "Google: Lyria 3 Clip Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "temperature": true,
        "tool_call": false
      },
      "google/lyria-3-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "google/lyria-3-pro-preview",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "audio",
            "text"
          ]
        },
        "name": "Google: Lyria 3 Pro Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "temperature": true,
        "tool_call": false
      },
      "gryphe/mythomax-l2-13b": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.06
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "gryphe/mythomax-l2-13b",
        "last_updated": "2024-04-25",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MythoMax 13B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-25",
        "temperature": true,
        "tool_call": false
      },
      "ibm-granite/granite-4.0-h-micro": {
        "attachment": false,
        "cost": {
          "input": 0.017,
          "output": 0.11
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "ibm-granite/granite-4.0-h-micro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "IBM: Granite 4.0 Micro",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-20",
        "temperature": true,
        "tool_call": false
      },
      "ibm-granite/granite-4.1-8b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.05,
          "output": 0.1
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "ibm-granite/granite-4.1-8b",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "IBM: Granite 4.1 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-30",
        "temperature": true,
        "tool_call": true
      },
      "inception/mercury-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "inception/mercury-2",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 128000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inception: Mercury 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "inclusionai/ling-2.6-1t",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "inclusionAI: Ling-2.6-1T",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-23",
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.016,
          "input": 0.08,
          "output": 0.24
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "inclusionai/ling-2.6-flash",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "inclusionAI: Ling-2.6 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ring-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.075,
          "output": 0.625
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "inclusionai/ring-2.6-1t",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "inclusionAI: Ring-2.6-1T",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-08",
        "temperature": true,
        "tool_call": true
      },
      "inflection/inflection-3-pi": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "inflection/inflection-3-pi",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 8000,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inflection: Inflection 3 Pi",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "temperature": true,
        "tool_call": false
      },
      "inflection/inflection-3-productivity": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "inflection/inflection-3-productivity",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 8000,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inflection: Inflection 3 Productivity",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "temperature": true,
        "tool_call": false
      },
      "kilo-auto/balanced": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "kilo-auto/balanced",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kilo Auto Balanced",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "kilo-auto/free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "kilo-auto/free",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kilo Auto Free",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "kilo-auto/frontier": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "kilo-auto/frontier",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kilo Auto Frontier",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "kilo-auto/small": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.4
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "kilo-auto/small",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kilo Auto Small",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "kwaipilot/kat-coder-pro-v2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "kwaipilot/kat-coder-pro-v2",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 256000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kwaipilot: KAT-Coder-Pro V2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-27",
        "temperature": true,
        "tool_call": true
      },
      "liquid/lfm-2-24b-a2b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.12
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "liquid/lfm-2-24b-a2b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LiquidAI: LFM2-24B-A2B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": false
      },
      "mancer/weaver": {
        "attachment": false,
        "cost": {
          "input": 0.75,
          "output": 1
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "mancer/weaver",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 8000,
          "output": 2000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mancer: Weaver (alpha)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-08-02",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.51,
          "output": 0.74
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3-70b-instruct",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 8192,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.04
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3-8b-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 8192,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-25",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3.1-70b-instruct",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3.1 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-16",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3.1-8b-instruct",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.049,
          "output": 0.049
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "id": "meta-llama/llama-3.2-11b-vision-instruct",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3.2 11B Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.2-1b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.027,
          "output": 0.2
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3.2-1b-instruct",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 60000,
          "output": 12000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3.2 1B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.051,
          "output": 0.34
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3.2-3b-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 80000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.32
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/llama-3.3-70b-instruct",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-01",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-4-maverick": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "id": "meta-llama/llama-4-maverick",
        "last_updated": "2025-12-24",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 4 Maverick",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-4-scout": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "output": 0.3
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "id": "meta-llama/llama-4-scout",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 327680,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama 4 Scout",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-guard-3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.06
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "id": "meta-llama/llama-guard-3-8b",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama Guard 3 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-guard-4-12b": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.18
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "id": "meta-llama/llama-guard-4-12b",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta: Llama Guard 4 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": false
      },
      "microsoft/phi-4": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.14
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "microsoft/phi-4",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Microsoft: Phi 4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": true,
        "tool_call": false
      },
      "microsoft/phi-4-mini-instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.08,
          "output": 0.35
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "microsoft/phi-4-mini-instruct",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Microsoft: Phi 4 Mini Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "microsoft/wizardlm-2-8x22b": {
        "attachment": false,
        "cost": {
          "input": 0.62,
          "output": 0.62
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "microsoft/wizardlm-2-8x22b",
        "last_updated": "2024-04-24",
        "limit": {
          "context": 65535,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "WizardLM-2 8x22B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-24",
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-01": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.1
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "id": "minimax/minimax-01",
        "last_updated": "2025-01-15",
        "limit": {
          "context": 1000192,
          "output": 1000192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax-01",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-15",
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-m1": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m1",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1000000,
          "output": 40000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax M1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.255,
          "output": 1
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2-her": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2-her",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 65536,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax M2-her",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-23",
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.27,
          "output": 0.95
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 196608,
          "output": 39322
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.029,
          "input": 0.25,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax-m2.7",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax: MiniMax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/codestral-2508": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "id": "mistralai/codestral-2508",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 256000,
          "output": 51200
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Codestral 2508",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/devstral-2512": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "id": "mistralai/devstral-2512",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Devstral 2 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-12",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/devstral-medium": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "id": "mistralai/devstral-medium",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Devstral Medium",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-10",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/devstral-small": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "id": "mistralai/devstral-small",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Devstral Small 1.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/ministral-14b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "id": "mistralai/ministral-14b-2512",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 262144,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Ministral 3 14B 2512",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/ministral-3b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "id": "mistralai/ministral-3b-2512",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Ministral 3 3B 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/ministral-8b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "id": "mistralai/ministral-8b-2512",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Ministral 3 8B 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-7b-instruct-v0.1": {
        "attachment": false,
        "cost": {
          "input": 0.11,
          "output": 0.19
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-7b-instruct-v0.1",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 2824,
          "output": 565
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral 7B Instruct v0.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "temperature": true,
        "tool_call": false
      },
      "mistralai/mistral-large": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "id": "mistralai/mistral-large",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 128000,
          "output": 25600
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-24",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-2407": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "id": "mistralai/mistral-large-2407",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 2407",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-19",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "id": "mistralai/mistral-large-2411",
        "last_updated": "2024-11-04",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 2411",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-24",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-2512": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "id": "mistralai/mistral-large-2512",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 262144,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Large 3 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-medium-3",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3-5": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-medium-3-5",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Medium 3.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3.1": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-medium-3.1",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Medium 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-12",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.04
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-nemo",
        "last_updated": "2024-07-30",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-saba": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-saba",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Saba",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-02-17",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-small-24b-instruct-2501": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.08
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-24b-instruct-2501",
        "last_updated": "2026-01-10",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Small 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-29",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-small-2603": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-2603",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Small 4",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-small-3.1-24b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.35,
          "output": 0.56
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-3.1-24b-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Small 3.1 24B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-17",
        "temperature": true,
        "tool_call": false
      },
      "mistralai/mistral-small-3.2-24b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.06,
          "output": 0.18
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-3.2-24b-instruct",
        "last_updated": "2025-06-20",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mistral Small 3.2 24B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mixtral-8x22b-instruct": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mixtral-8x22b-instruct",
        "last_updated": "2024-04-17",
        "limit": {
          "context": 65536,
          "output": 13108
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mixtral 8x22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-17",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/pixtral-large-2411": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral vision-language model for image understanding and multimodal chat",
        "id": "mistralai/pixtral-large-2411",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Pixtral Large 2411",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-19",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/voxtral-small-24b-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/voxtral-small-24b-2507",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 32000,
          "output": 6400
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Voxtral Small 24B 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "moonshotai/kimi-k2",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131000,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI: Kimi K2 0711",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.4,
          "output": 2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "moonshotai/kimi-k2-0905",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI: Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.47,
          "output": 2
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "moonshotai/kimi-k2-thinking",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI: Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "input": 0.45,
          "output": 2.2
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "moonshotai/kimi-k2.5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65535
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI: Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.375,
          "input": 0.75,
          "output": 3.5
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "moonshotai/kimi-k2.6",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 262144,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI: Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "morph/morph-v3-fast": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "morph/morph-v3-fast",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 81920,
          "output": 38000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph: Morph V3 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": true,
        "tool_call": false
      },
      "morph/morph-v3-large": {
        "attachment": false,
        "cost": {
          "input": 0.9,
          "output": 1.9
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "morph/morph-v3-large",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph: Morph V3 Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": true,
        "tool_call": false
      },
      "nex-agi/deepseek-v3.1-nex-n1": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "nex-agi/deepseek-v3.1-nex-n1",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131072,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nex AGI: DeepSeek V3.1 Nex N1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "nousresearch/hermes-2-pro-llama-3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.14
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "nousresearch/hermes-2-pro-llama-3-8b",
        "last_updated": "2024-06-27",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NousResearch: Hermes 2 Pro - Llama-3 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-05-27",
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-3-llama-3.1-405b": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "nousresearch/hermes-3-llama-3.1-405b",
        "last_updated": "2024-08-16",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nous: Hermes 3 405B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-16",
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-3-llama-3.1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "nousresearch/hermes-3-llama-3.1-70b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nous: Hermes 3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-18",
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-4-405b": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "nousresearch/hermes-4-405b",
        "last_updated": "2025-08-25",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nous: Hermes 4 405B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-25",
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-4-70b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.055,
          "input": 0.13,
          "output": 0.4
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "nousresearch/hermes-4-70b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nous: Hermes 4 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-25",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/llama-3.3-nemotron-super-49b-v1.5": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5",
        "last_updated": "2025-03-16",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA: Llama 3.3 Nemotron Super 49B V1.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-03-16",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-nano-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-30b-a3b",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 262144,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA: Nemotron 3 Nano 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2024-12",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "audio",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA: Nemotron 3 Nano Omni (free)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-28",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.1,
          "output": 0.5
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA: Nemotron 3 Super",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b:free",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA: Nemotron 3 Super (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-12",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-nano-9b-v2": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.16
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-nano-9b-v2",
        "last_updated": "2025-08-18",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA: Nemotron Nano 9B V2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-3.5-turbo",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-3.5 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-0613": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-3.5-turbo-0613",
        "last_updated": "2023-06-13",
        "limit": {
          "context": 4095,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-3.5 Turbo (older v0613)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-06-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-16k": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-3.5-turbo-16k",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-3.5 Turbo 16k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-08-28",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-3.5-turbo-instruct",
        "last_updated": "2023-09-21",
        "limit": {
          "context": 4095,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-3.5 Turbo Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4": {
        "attachment": false,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 8191,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-0314": {
        "attachment": false,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4-0314",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 8191,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4 (older v0314)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-05-28",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-1106-preview": {
        "attachment": false,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4-1106-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4 Turbo (older v1106)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4-turbo",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-turbo-preview": {
        "attachment": false,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4-turbo-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4 Turbo Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4.1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4.1-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4.1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4.1-nano",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4.1 Nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4o",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-05-13": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 15
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4o-2024-05-13",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o (2024-05-13)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-08-06": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4o-2024-08-06",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o (2024-08-06)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-06",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4o-2024-11-20",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-audio-preview": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "openai/gpt-4o-audio-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "audio",
            "text"
          ],
          "output": [
            "audio",
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o Audio",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-15",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4o-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini-2024-07-18": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4o-mini-2024-07-18",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o-mini (2024-07-18)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini-search-preview": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-4o-mini-search-preview",
        "last_updated": "2025-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o-mini Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01",
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-4o-search-preview": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-4o-search-preview",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-4o Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-13",
        "tool_call": false
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-5-chat",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-5-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5-codex",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-image": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-5-image",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Image",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-image-mini": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 2
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-5-image-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Image Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-16",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5-nano",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5-pro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5.1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-5.1-chat",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.1-codex",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.1-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.1-codex-max",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.1-Codex-Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.1-codex-mini",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.1-Codex-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5.2",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-5.2-chat",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.2-codex",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.2-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-14",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.2-pro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-chat": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-5.3-chat",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.3 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-04",
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.3-codex",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.3-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-25",
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.4",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-06",
        "tool_call": true
      },
      "openai/gpt-5.4-image-2": {
        "attachment": true,
        "cost": {
          "cache_read": 2,
          "input": 8,
          "output": 15
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-5.4-image-2",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.4 Image 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-mini",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-nano",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.4 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.4-pro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-06",
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.5",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.5-pro",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-audio": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "openai/gpt-audio",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "audio",
            "text"
          ],
          "output": [
            "audio",
            "text"
          ]
        },
        "name": "OpenAI: GPT Audio",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-20",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-audio-mini": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "openai/gpt-audio-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "audio",
            "text"
          ],
          "output": [
            "audio",
            "text"
          ]
        },
        "name": "OpenAI: GPT Audio Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-20",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-chat-latest",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT Chat Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.039,
          "output": 0.19
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: gpt-oss-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.14
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 26215
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: gpt-oss-20b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-safeguard-20b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.037,
          "input": 0.075,
          "output": 0.3
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "id": "openai/gpt-oss-safeguard-20b",
        "last_updated": "2025-10-29",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: gpt-oss-safeguard-20b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-29",
        "temperature": true,
        "tool_call": true
      },
      "openai/o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-05",
        "temperature": false,
        "tool_call": true
      },
      "openai/o1-pro": {
        "attachment": true,
        "cost": {
          "input": 150,
          "output": 600
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o1-pro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o1-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-03-19",
        "temperature": false,
        "tool_call": false
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o3",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 2.5,
          "input": 10,
          "output": 40
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "id": "openai/o3-deep-research",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o3 Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2024-06-26",
        "temperature": true,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o3-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o3 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-20",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini-high": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o3-mini-high",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o3 Mini High",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-31",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-pro": {
        "attachment": true,
        "cost": {
          "input": 20,
          "output": 80
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o3-pro",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o3 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o4-mini",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "id": "openai/o4-mini-deep-research",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o4 Mini Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2024-06-26",
        "temperature": true,
        "tool_call": true
      },
      "openai/o4-mini-high": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "openai/o4-mini-high",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: o4 Mini High",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-04-17",
        "tool_call": true
      },
      "openrouter/auto": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openrouter/auto",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 2000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "audio",
            "image",
            "pdf",
            "text",
            "video"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Auto Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "openrouter/bodybuilder": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Preview model for early access evaluation, prototyping, and compatibility testing",
        "id": "openrouter/bodybuilder",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Body Builder (beta)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-15",
        "status": "beta",
        "tool_call": false
      },
      "openrouter/free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "openrouter/free",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Free Models Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "openrouter/owl-alpha": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "openrouter/owl-alpha",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 1048756,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Owl Alpha",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "status": "alpha",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openrouter/pareto-code": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "openrouter/pareto-code",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 200000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pareto Code Router",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "perceptron/perceptron-mk1": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "perceptron/perceptron-mk1",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perceptron: Perceptron Mk1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "id": "perplexity/sonar",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 127072,
          "output": 25415
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity: Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-deep-research": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "id": "perplexity/sonar-deep-research",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 128000,
          "output": 25600
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity: Sonar Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-27",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-pro": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "id": "perplexity/sonar-pro",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 200000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity: Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-pro-search": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "id": "perplexity/sonar-pro-search",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 200000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity: Sonar Pro Search",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-31",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-reasoning-pro": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Web-grounded reasoning model for multi-step research and cited answers",
        "id": "perplexity/sonar-reasoning-pro",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 25600
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity: Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "poolside/laguna-m.1:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "id": "poolside/laguna-m.1:free",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Poolside: Laguna M.1 (free)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-xs.2:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "id": "poolside/laguna-xs.2:free",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Poolside: Laguna XS.2 (free)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "temperature": true,
        "tool_call": true
      },
      "prime-intellect/intellect-3": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.1
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "prime-intellect/intellect-3",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Prime Intellect: INTELLECT-3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-26",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.39
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen-2.5-72b-instruct",
        "last_updated": "2026-01-10",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen-2.5-7b-instruct",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 32768,
          "output": 6554
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen2.5 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-coder-32b-instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.2,
          "output": 0.2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen-2.5-coder-32b-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 Coder 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-11",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 1.2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen-plus",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen-Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-plus-2025-07-28": {
        "attachment": false,
        "cost": {
          "input": 0.26,
          "output": 0.78
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen-plus-2025-07-28",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen Plus 0728",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-plus-2025-07-28:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.26,
          "output": 0.78
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen-plus-2025-07-28:thinking",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen Plus 0728 (thinking)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen2.5-vl-72b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.8,
          "output": 0.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen2.5-vl-72b-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen2.5 VL 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-02-01",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-14b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.06,
          "output": 0.24
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-14b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 14B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.455,
          "output": 1.82
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-235b-a22b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 235B A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-2507": {
        "attachment": false,
        "cost": {
          "input": 0.071,
          "output": 0.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-235b-a22b-2507",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.11,
          "output": 0.6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-235b-a22b-thinking-2507",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 235B A22B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.08,
          "output": 0.28
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-30b-a3b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-instruct-2507": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "input": 0.09,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-30b-a3b-instruct-2507",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 30B A3B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.051,
          "output": 0.34
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-30b-a3b-thinking-2507",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 32768,
          "output": 6554
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 30B A3B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-29",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-32b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "input": 0.08,
          "output": 0.24
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-32b",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-8b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-8b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 40960,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 8B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder": {
        "attachment": false,
        "cost": {
          "cache_read": 0.022,
          "input": 0.22,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Coder 480B A35B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.27
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-30b-a3b-instruct",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 160000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Coder 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.195,
          "output": 0.975
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-flash",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Coder Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "cache_read": 0.035,
          "input": 0.12,
          "output": 0.75
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-next",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Coder Next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.65,
          "output": 3.25
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-plus",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Coder Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 6
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "id": "qwen/qwen3-max",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.78,
          "output": 3.9
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-max-thinking",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Max Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 1.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-next-80b-a3b-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Next 80B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.0975,
          "output": 0.78
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-next-80b-a3b-thinking",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 Next 80B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-235b-a22b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.2,
          "output": 0.88
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-235b-a22b-instruct",
        "last_updated": "2026-01-10",
        "limit": {
          "context": 262144,
          "output": 52429
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 235B A22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-235b-a22b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.26,
          "output": 2.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-235b-a22b-thinking",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 235B A22B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-24",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-30b-a3b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-30b-a3b-instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-30b-a3b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 1.56
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-30b-a3b-thinking",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 30B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-11",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-32b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.104,
          "output": 0.416
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-32b-instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 32B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-8b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "output": 0.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-8b-instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-8b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.117,
          "output": 1.365
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-8b-thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3 VL 8B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0.26,
          "output": 2.08
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-122b-a10b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5-122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.195,
          "output": 1.56
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-27b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5-27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.1625,
          "output": 1.3
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-35b-a3b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5-35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.39,
          "output": 2.34
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-397b-a17b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5 397B A17B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-9b": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.15
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-9b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5-9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-flash-02-23": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-flash-02-23",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-plus-02-15": {
        "attachment": true,
        "cost": {
          "input": 0.26,
          "output": 1.56
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-plus-02-15",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5 Plus 2026-02-15",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-15",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-plus-20260420": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-plus-20260420",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.5 Plus 2026-04-20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.325,
          "output": 3.25
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.6-27b",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.6 27B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1612,
          "input": 0.1612,
          "output": 0.96525
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.6-35b-a3b",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.6 35B A3B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_write": 0.3125,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.6-flash",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "cache_write": 1.3,
          "input": 1.04,
          "output": 6.24
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "id": "qwen/qwen3.6-max-preview",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0325,
          "cache_write": 0.40625,
          "input": 0.325,
          "output": 1.95
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.6-plus",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-26",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1625,
          "cache_write": 2.03125,
          "input": 1.625,
          "output": 4.875
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "id": "qwen/qwen3.7-max",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-26",
        "temperature": true,
        "tool_call": true
      },
      "rekaai/reka-edge": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "rekaai/reka-edge",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Reka Edge",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "rekaai/reka-flash-3": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "rekaai/reka-flash-3",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Reka Flash 3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-12",
        "temperature": true,
        "tool_call": false
      },
      "relace/relace-apply-3": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 1.25
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "relace/relace-apply-3",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Relace: Relace Apply 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-26",
        "tool_call": false
      },
      "relace/relace-search": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "relace/relace-search",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Relace: Relace Search",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-09",
        "temperature": true,
        "tool_call": true
      },
      "sao10k/l3-euryale-70b": {
        "attachment": false,
        "cost": {
          "input": 1.48,
          "output": 1.48
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "sao10k/l3-euryale-70b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10k: Llama 3 Euryale 70B v2.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-18",
        "temperature": true,
        "tool_call": true
      },
      "sao10k/l3-lunaris-8b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "sao10k/l3-lunaris-8b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10K: Llama 3 8B Lunaris",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-13",
        "temperature": true,
        "tool_call": false
      },
      "sao10k/l3.1-70b-hanami-x1": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 3
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "sao10k/l3.1-70b-hanami-x1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 16000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10K: Llama 3.1 70B Hanami x1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-08",
        "temperature": true,
        "tool_call": false
      },
      "sao10k/l3.1-euryale-70b": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 0.85
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "sao10k/l3.1-euryale-70b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10K: Llama 3.1 Euryale 70B v2.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-28",
        "temperature": true,
        "tool_call": true
      },
      "sao10k/l3.3-euryale-70b": {
        "attachment": false,
        "cost": {
          "input": 0.65,
          "output": 0.75
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "sao10k/l3.3-euryale-70b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10K: Llama 3.3 Euryale 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-18",
        "temperature": true,
        "tool_call": false
      },
      "stealth/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4,
          "cache_write": 5,
          "input": 4,
          "output": 20
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "stealth/claude-opus-4.6",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Stealth: Claude Opus 4.6 (20% off)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "stealth/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4,
          "cache_write": 5,
          "input": 4,
          "output": 20
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "stealth/claude-opus-4.7",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Stealth: Claude Opus 4.7 (20% off)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "stealth/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "cache_write": 3,
          "input": 2.4,
          "output": 12
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "stealth/claude-sonnet-4.6",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Stealth: Claude Sonnet 4.6 (20% off)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun/step-3.5-flash",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "StepFun: Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      },
      "switchpoint/router": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 3.4
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "switchpoint/router",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Switchpoint Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-12",
        "temperature": true,
        "tool_call": false
      },
      "tencent/hunyuan-a13b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "id": "tencent/hunyuan-a13b-instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tencent: Hunyuan A13B Instruct",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": false
      },
      "tencent/hy3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.029,
          "input": 0.066,
          "output": 0.26
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "id": "tencent/hy3-preview",
        "last_updated": "2026-05-16",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tencent: Hy3 Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "thedrummer/cydonia-24b-v4.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/cydonia-24b-v4.1",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TheDrummer: Cydonia 24B V4.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-27",
        "temperature": true,
        "tool_call": false
      },
      "thedrummer/rocinante-12b": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.43
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/rocinante-12b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TheDrummer: Rocinante 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-30",
        "temperature": true,
        "tool_call": true
      },
      "thedrummer/skyfall-36b-v2": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 0.8
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/skyfall-36b-v2",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TheDrummer: Skyfall 36B V2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-11",
        "temperature": true,
        "tool_call": false
      },
      "thedrummer/unslopnemo-12b": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/unslopnemo-12b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TheDrummer: UnslopNemo 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-09",
        "temperature": true,
        "tool_call": true
      },
      "undi95/remm-slerp-l2-13b": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 0.65
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "undi95/remm-slerp-l2-13b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 6144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ReMM SLERP 13B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-07-22",
        "temperature": true,
        "tool_call": false
      },
      "upstage/solar-pro-3": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "upstage/solar-pro-3",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Upstage: Solar Pro 3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "writer/palmyra-x5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 6
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "writer/palmyra-x5",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 1040000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Writer: Palmyra X5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": false
      },
      "x-ai/grok-4.20": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4.20",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI: Grok 4.20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-31",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.20-multi-agent": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4.20-multi-agent",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "image",
            "pdf",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI: Grok 4.20 Multi-Agent",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-31",
        "temperature": true,
        "tool_call": false
      },
      "x-ai/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4.3",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI: Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-01",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 2
        },
        "description": "Grok coding model for agentic engineering, edits, and codebase workflows",
        "id": "x-ai/grok-build-0.1",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "xAI: Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-20",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.045,
          "input": 0.09,
          "output": 0.29
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash",
        "knowledge": "2024-12-01",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi: MiMo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 2
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-omni",
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi: MiMo-V2-Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-pro",
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi: MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "context_over_200k": {
            "cache_read": 0.16,
            "input": 0.8,
            "output": 4
          },
          "input": 0.4,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.16,
              "input": 0.8,
              "output": 4,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "xiaomi/mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi: MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "xiaomi/mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi: MiMo V2.5 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4-32b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4-32b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4 32B ",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.13,
          "output": 0.85
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-4.5-air",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.5 Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-4.5v",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 0.39,
          "output": 1.9
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.6",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 204800,
          "output": 204800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6v": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-4.6v",
        "last_updated": "2026-01-10",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.38,
          "output": 1.98
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.7",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 202752,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.06,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-4.7-flash",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 202752,
          "output": 40551
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 4.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 2.3
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-5",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-5-turbo",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 5 Turbo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.1": {
        "attachment": false,
        "cost": {
          "input": 1.26,
          "output": 3.96
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-5.1",
        "last_updated": "2026-03-27",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-27",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-5v-turbo",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z.ai: GLM 5V Turbo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "~anthropic/claude-haiku-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "~anthropic/claude-haiku-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Haiku Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "~anthropic/claude-opus-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "~anthropic/claude-opus-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Opus Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "~anthropic/claude-sonnet-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "~anthropic/claude-sonnet-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic: Claude Sonnet Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "~google/gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.08333333333333334,
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "~google/gemini-flash-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "~google/gemini-pro-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.375,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "~google/gemini-pro-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google: Gemini Pro Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "~moonshotai/kimi-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.14,
          "input": 0.74,
          "output": 3.49
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "~moonshotai/kimi-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262142,
          "output": 262142
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI: Kimi Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-27",
        "temperature": true,
        "tool_call": true
      },
      "~openai/gpt-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "~openai/gpt-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "temperature": false,
        "tool_call": true
      },
      "~openai/gpt-mini-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "~openai/gpt-mini-latest",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT Mini Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Kilo Gateway",
    "npm": "@ai-sdk/openai-compatible"
  },
  "kimi-for-coding": {
    "api": "https://api.kimi.com/coding/v1",
    "doc": "https://www.kimi.com/code/docs/en/third-party-tools/other-coding-agents.html",
    "env": [
      "KIMI_API_KEY"
    ],
    "id": "kimi-for-coding",
    "models": {
      "k2p5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "k2p5",
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "k2p6": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "k2p6",
        "knowledge": "2025-01",
        "last_updated": "2026-04",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "k2p7": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "k2p7",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "knowledge": "2025-07",
        "last_updated": "2025-12",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Kimi For Coding",
    "npm": "@ai-sdk/anthropic"
  },
  "kuae-cloud-coding-plan": {
    "api": "https://coding-plan-endpoint.kuaecloud.net/v1",
    "doc": "https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/",
    "env": [
      "KUAE_API_KEY"
    ],
    "id": "kuae-cloud-coding-plan",
    "models": {
      "GLM-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "GLM-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "KUAE Cloud Coding Plan",
    "npm": "@ai-sdk/openai-compatible"
  },
  "lilac": {
    "api": "https://api.getlilac.com/v1",
    "doc": "https://docs.getlilac.com/inference/models",
    "env": [
      "LILAC_API_KEY"
    ],
    "id": "lilac",
    "models": {
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.11,
          "output": 0.35
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262100,
          "output": 262100
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimaxai/minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.28,
          "output": 1.1
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax-m3",
        "id": "minimaxai/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 1048576,
          "output": 1048576
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 0.7,
          "output": 3.5
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.27,
          "input": 0.9,
          "output": 3
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 524288,
          "output": 524288
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Lilac",
    "npm": "@ai-sdk/openai-compatible"
  },
  "llama": {
    "api": "https://api.llama.com/compat/v1/",
    "doc": "https://llama.developer.meta.com/docs/models",
    "env": [
      "LLAMA_API_KEY"
    ],
    "id": "llama",
    "models": {
      "cerebras-llama-4-maverick-17b-128e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "cerebras-llama-4-maverick-17b-128e-instruct",
        "knowledge": "2025-01",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cerebras-Llama-4-Maverick-17B-128E-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "cerebras-llama-4-scout-17b-16e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "cerebras-llama-4-scout-17b-16e-instruct",
        "knowledge": "2025-01",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cerebras-Llama-4-Scout-17B-16E-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "groq-llama-4-maverick-17b-128e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "groq-llama-4-maverick-17b-128e-instruct",
        "knowledge": "2025-01",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Groq-Llama-4-Maverick-17B-128E-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-8b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-8B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-maverick-17b-128e-instruct-fp8": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "llama-4-maverick-17b-128e-instruct-fp8",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-4-Maverick-17B-128E-Instruct-FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-scout-17b-16e-instruct-fp8": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "llama-4-scout-17b-16e-instruct-fp8",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-4-Scout-17B-16E-Instruct-FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Llama",
    "npm": "@ai-sdk/openai-compatible"
  },
  "llmgateway": {
    "api": "https://api.llmgateway.io/v1",
    "doc": "https://llmgateway.io/docs",
    "env": [
      "LLMGATEWAY_API_KEY"
    ],
    "id": "llmgateway",
    "models": {
      "auto": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "auto",
        "id": "auto",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto Route",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-3-7-sonnet": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude",
        "id": "claude-3-7-sonnet",
        "last_updated": "2025-02-24",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.7 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 8191,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-02-24",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-3-7-sonnet-20250219": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-3-7-sonnet-20250219",
        "knowledge": "2024-10-31",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 8191,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": true
      },
      "claude-3-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude",
        "id": "claude-3-opus",
        "last_updated": "2024-03-04",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3 Opus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-haiku-4-5-free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-free",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1-20250805",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5-20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "codestral-2508": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "mistral",
        "id": "codestral-2508",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "custom": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "auto",
        "id": "custom",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Custom Model",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.112,
          "input": 0.56,
          "output": 1.68
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.1",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 0.26,
          "output": 0.38
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163840,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1050000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1050000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "devstral-2512": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes",
        "family": "devstral",
        "id": "devstral-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "devstral-small-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "devstral-small-2507",
        "knowledge": "2025-05",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 131072,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-10",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "fugu-ultra": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Quality-first multi-agent model for hard research, analysis, and competitions",
        "id": "fugu-ultra",
        "last_updated": "2026-06-22",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fugu Ultra",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 24576,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite-preview-09-2025": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini",
        "id": "gemini-2.5-flash-lite-preview-09-2025",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 1048576
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite Preview (09-2025)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.08333,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "cache_write": 0.08333,
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-pro-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini",
        "id": "gemini-pro-latest",
        "last_updated": "2026-02-27",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Pro Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.07,
          "output": 0.34
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.38
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4-32b-0414-128k": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4-32b-0414-128k",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4 32B (0414-128k)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131000,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0,
          "input": 0.13,
          "output": 0.85
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131000,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-airx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.22,
          "input": 1.1,
          "output": 4.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.5-airx",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5 AirX",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-x": {
        "attachment": false,
        "cost": {
          "cache_read": 0.45,
          "input": 2.2,
          "output": 8.9
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.5-x",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5 X",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.5v",
        "knowledge": "2025-04",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.431,
          "output": 2.007
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6v-flashx": {
        "attachment": true,
        "cost": {
          "cache_read": 0.004,
          "input": 0.04,
          "output": 0.4
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.6v-flashx",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V FlashX",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.19,
          "cache_write": 0,
          "input": 0.38,
          "output": 1.98
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0,
          "input": 0.06,
          "output": 0.4
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm-flash",
        "id": "glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0,
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "glm-4.7-flashx",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-FlashX",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.144,
          "cache_write": 0,
          "input": 0.72,
          "output": 2.3
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 203000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.173,
          "cache_write": 0,
          "input": 0.931,
          "output": 2.93
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.234,
          "cache_write": 0,
          "input": 1.26,
          "output": 3.96
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo",
        "knowledge": "2021-09-01",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gpt-4": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini-search-preview": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4o-mini-search-preview",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Mini Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gpt-4o-search-preview": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o-search-preview",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Codex GPT for repository edits, code review, and practical software agents",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.2-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows",
        "family": "gpt-pro",
        "id": "gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5.3-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt-pro",
        "id": "gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32766
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.15
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32766
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-non-reasoning",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-1-fast-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "grok-4-1-fast-reasoning",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20-beta-0309-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 2,
          "output": 6,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-20-beta-0309-non-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20-beta-0309-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 2,
          "output": 6,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use",
        "family": "grok",
        "id": "grok-4-20-beta-0309-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-20-non-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use",
        "family": "grok",
        "id": "grok-4-20-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3125,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "grok-4-3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-build-0-1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 4
          },
          "input": 1,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 4,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "grok-build-0-1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 0.574,
          "output": 2.294
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.574,
          "output": 2.294
        },
        "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.225,
          "input": 0.405,
          "output": 1.98
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 2.2
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.7-code-highspeed": {
        "attachment": true,
        "cost": {
          "cache_read": 0.38,
          "input": 1.9,
          "output": 8
        },
        "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code Highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "llama-3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.51,
          "output": 0.74
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3-70b-instruct",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3-8b-instruct",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.1-70b-instruct",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "status": "beta",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "llama-3.1-nemotron-ultra-253b": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 1.8
        },
        "description": "Flagship Nemotron model for high-throughput reasoning and complex agents",
        "family": "nemotron",
        "id": "llama-3.1-nemotron-ultra-253b",
        "last_updated": "2025-04-07",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 Nemotron Ultra 253B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "llama-3.2-11b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.33
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.2-11b-instruct",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.2-3b-instruct",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 32768,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "llama-3.3-70b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "llama-4-maverick-17b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.27,
          "output": 0.85
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "llama-4-maverick-17b-instruct",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1048576,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "llama-4-scout-17b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.59
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "llama-4-scout-17b-instruct",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 131072,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 2
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo",
        "id": "mimo-v2-omni",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 256000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Omni",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "context_over_200k": {
            "cache_read": 0.16,
            "input": 0.8,
            "output": 4
          },
          "input": 0.14,
          "output": 0.28,
          "tiers": [
            {
              "cache_read": 0.16,
              "input": 0.8,
              "output": 4,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0036,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 0.435,
          "output": 0.87,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.2,
          "output": 1
        },
        "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows",
        "family": "minimax",
        "id": "minimax-m2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.1": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1.1
        },
        "description": "Earlier MiniMax agent model for practical coding and productivity tasks",
        "family": "minimax",
        "id": "minimax-m2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.1-lightning": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.48
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "minimax-m2.1-lightning",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 196608,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1 Lightning",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 228700,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "minimax-m2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax-m2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 2.4
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "minimax-text-01": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.1
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-text-01",
        "last_updated": "2025-01-15",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax Text 01",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "ministral-14b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "mistral",
        "id": "ministral-14b-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 14B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "ministral-3b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "mistral",
        "id": "ministral-3b-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "ministral-8b-2512": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "mistral",
        "id": "ministral-8b-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "mistral-large-2512": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "mistral-large-2512",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-latest": {
        "attachment": true,
        "cost": {
          "input": 4,
          "output": 12
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-latest",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 128000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-2506": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-2506",
        "knowledge": "2025-03",
        "last_updated": "2025-06-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-ultra-550b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.5,
          "output": 2.5
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nemotron-3-ultra-550b",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra 550B A55B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-04",
        "temperature": true,
        "tool_call": true
      },
      "o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "pixtral-large-latest": {
        "attachment": true,
        "cost": {
          "input": 4,
          "output": 12
        },
        "description": "Mistral's larger vision model for document-heavy image understanding and chat",
        "family": "pixtral",
        "id": "pixtral-large-latest",
        "knowledge": "2024-11",
        "last_updated": "2024-11-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-coder-plus": {
        "attachment": false,
        "cost": {
          "input": 0.502,
          "output": 1.004
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen-coder-plus",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Coder Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.0625,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-flash",
        "knowledge": "2024-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen-max": {
        "attachment": false,
        "cost": {
          "input": 1.6,
          "output": 6.4
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen-max",
        "knowledge": "2024-04",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-03",
        "temperature": true,
        "tool_call": true
      },
      "qwen-max-latest": {
        "attachment": true,
        "cost": {
          "input": 1.6,
          "output": 6.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-max-latest",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Max Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen-omni-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen-omni-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-03-26",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen-Omni Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-19",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 0.5,
          "input": 0.4,
          "output": 1.2,
          "reasoning": 4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen-plus-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 0.5,
          "input": 0.4,
          "output": 1.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-plus-latest",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 1000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2,
          "reasoning": 0.5
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen-turbo",
        "knowledge": "2024-04",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-max": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-vl-max",
        "knowledge": "2024-04",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-08",
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-plus": {
        "attachment": false,
        "cost": {
          "input": 0.21,
          "output": 0.64
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen-vl-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-08-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-VL Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-vl-32b-instruct": {
        "attachment": true,
        "cost": {
          "input": 1.4,
          "output": 4.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen2-5-vl-32b-instruct",
        "last_updated": "2025-03-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 VL 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen2-5-vl-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen2-5-vl-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-09",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b-fp8",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 40960,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.58
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b-instruct-2507",
        "last_updated": "2025-07-08",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct (2507)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen3-235b-a22b-thinking-2507",
        "last_updated": "2025-07-08",
        "limit": {
          "context": 262000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking (2507)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-30b-a3b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-30b-a3b-instruct-2507",
        "last_updated": "2025-07-08",
        "limit": {
          "context": 262000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct (2507)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3,
          "reasoning": 8.4
        },
        "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
        "family": "qwen",
        "id": "qwen3-32b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-4b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.03
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-4b-fp8",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 4B FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.27
        },
        "description": "Smaller Qwen coder for efficient local agents and repo-level fixes",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.3
        },
        "description": "Open Qwen coding heavyweight for repository reasoning and agentic engineering",
        "family": "qwen",
        "id": "qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 480B-A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-flash",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.108,
          "output": 0.675
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-next",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 1.2,
          "cache_write": 7.5,
          "input": 6,
          "output": 60
        },
        "description": "Hosted Qwen coder for software agents, repo edits, and long-context code",
        "family": "qwen",
        "id": "qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.6,
          "cache_write": 3.75,
          "input": 0.845,
          "output": 3.38
        },
        "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max-2026-01-23": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "cache_write": 1.5,
          "input": 1.2,
          "output": 6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-max-2026-01-23",
        "last_updated": "2026-01-23",
        "limit": {
          "context": 262144,
          "output": 32800
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max (2026-01-23)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.2
        },
        "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents",
        "family": "qwen",
        "id": "qwen3-next-80b-a3b-thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B (Thinking)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-235b-a22b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-235b-a22b-instruct",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-235b-a22b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-235b-a22b-thinking",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-30b-a3b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.7
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-30b-a3b-instruct",
        "last_updated": "2025-10-02",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-30b-a3b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-30b-a3b-thinking",
        "last_updated": "2025-10-02",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 30B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-8b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "output": 0.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-8b-instruct",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen3-vl-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-flash",
        "last_updated": "2025-10-09",
        "limit": {
          "context": 262144,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "cache_write": 0.25,
          "input": 0.2,
          "output": 1.6,
          "reasoning": 4.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-vl-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-9b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.15
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3.5-9b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.248,
          "output": 1.485
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "qwen3.6-35b-a3b",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "cache_write": 1.625,
          "input": 1.3,
          "output": 7.8
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen3.6-max-preview",
        "knowledge": "2025-04",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 3.125,
          "input": 1.25,
          "output": 3.75
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 0.5,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen35-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen35-397b-a17b",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwq-plus": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 2.4
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwq-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-03-05",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwQ Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-05",
        "temperature": true,
        "tool_call": true
      },
      "seed-1-6-250615": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "seed-1-6-250615",
        "last_updated": "2025-06-25",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.6 (250615)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "seed-1-6-250915": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "seed-1-6-250915",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.6 (250915)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "seed-1-6-flash-250715": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.07,
          "output": 0.3
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "seed-1-6-flash-250715",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.6 Flash (250715)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-07-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "seed-1-8-251228": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "seed-1-8-251228",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.8 (251228)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "sonar": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Fast web-grounded Sonar for current answers, citations, and lightweight retrieval",
        "family": "sonar",
        "id": "sonar",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 130000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "sonar-pro": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Deeper Sonar search model with broader retrieval and stronger synthesis",
        "family": "sonar-pro",
        "id": "sonar-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "sonar-reasoning-pro": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Web-grounded Sonar for multi-step research questions that need cited reasoning",
        "family": "sonar-reasoning",
        "id": "sonar-reasoning-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "LLM Gateway",
    "npm": "@ai-sdk/openai-compatible"
  },
  "llmtr": {
    "api": "https://llmtr.com/v1",
    "doc": "https://llmtr.com/docs",
    "env": [
      "LLMTR_API_KEY"
    ],
    "id": "llmtr",
    "models": {
      "gemma-4": {
        "attachment": false,
        "cost": {
          "input": 5,
          "output": 10
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "gemma-4",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": false
      },
      "magibu-11b-v8": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "magibu-11b-v8",
        "last_updated": "2026-06-05",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magibu 11B v8",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-05",
        "temperature": true,
        "tool_call": false
      },
      "medgemma-4b": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 5
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "medgemma-4b",
        "last_updated": "2026-04-26",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MedGemma 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-26",
        "temperature": true,
        "tool_call": false
      },
      "qwen3-6-35b": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 10
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "qwen3-6-35b",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 16384,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "sincap": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "sincap",
        "last_updated": "2026-05-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sincap",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-05",
        "temperature": true,
        "tool_call": false
      },
      "trendyol-7b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "trendyol-7b",
        "last_updated": "2026-06-06",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trendyol 7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-06-06",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "LLMTR",
    "npm": "@ai-sdk/openai-compatible"
  },
  "lmstudio": {
    "api": "http://127.0.0.1:1234/v1",
    "doc": "https://lmstudio.ai/models",
    "env": [
      "LMSTUDIO_API_KEY"
    ],
    "id": "lmstudio",
    "models": {
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-2507": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-30b-a3b-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-30b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-30b",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "LMStudio",
    "npm": "@ai-sdk/openai-compatible"
  },
  "lucidquery": {
    "api": "https://api.lucidquery.com/v1",
    "doc": "https://lucidquery.com/docs",
    "env": [
      "LUCIDQUERY_API_KEY"
    ],
    "id": "lucidquery",
    "models": {
      "lucidnova-rf1-100b": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "nova",
        "id": "lucidnova-rf1-100b",
        "knowledge": "2025-09-16",
        "last_updated": "2025-09-10",
        "limit": {
          "context": 120000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LucidNova RF1 100B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-28",
        "temperature": false,
        "tool_call": true
      },
      "lucidquery-agi-01-frontier": {
        "attachment": true,
        "cost": {
          "input": 4.5,
          "output": 22
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "agi",
        "id": "lucidquery-agi-01-frontier",
        "knowledge": "2026-06-05",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 300000,
          "output": 120000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AGI-01 Frontier",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-16",
        "temperature": true,
        "tool_call": true
      },
      "lucidquery-agi-01-swift": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 15
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "agi",
        "id": "lucidquery-agi-01-swift",
        "knowledge": "2026-06-05",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 300000,
          "output": 120000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AGI-01 Swift",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-16",
        "temperature": true,
        "tool_call": true
      },
      "lucidquery-nexus-coder": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 5
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "lucid",
        "id": "lucidquery-nexus-coder",
        "knowledge": "2025-08-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 250000,
          "output": 60000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LucidQuery Nexus Coder",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-01",
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "LucidQuery",
    "npm": "@ai-sdk/openai-compatible"
  },
  "meganova": {
    "api": "https://api.meganova.ai/v1",
    "doc": "https://docs.meganova.ai",
    "env": [
      "MEGANOVA_API_KEY"
    ],
    "id": "meganova",
    "models": {
      "MiniMaxAI/MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-12-23",
        "limit": {
          "context": 196608,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-VL-32B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-VL-32B-Instruct",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 VL 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-Plus": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2.4,
          "reasoning": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-Plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02",
        "temperature": true,
        "tool_call": true
      },
      "XiaomiMiMo/MiMo-V2-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "XiaomiMiMo/MiMo-V2-Flash",
        "knowledge": "2024-12-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 262144,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.15
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1-0528",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-V3-0324": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.88
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3-0324",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1",
        "last_updated": "2025-08-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "input": 0.26,
          "output": 0.38
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "last_updated": "2025-12-03",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2-Exp": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.4
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2-Exp",
        "last_updated": "2025-10-10",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Exp",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Mistral-Nemo-Instruct-2407": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.04
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "mistralai/Mistral-Nemo-Instruct-2407",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo Instruct 2407",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/Mistral-Small-3.2-24B-Instruct-2506": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistralai/Mistral-Small-3.2-24B-Instruct-2506",
        "knowledge": "2024-10",
        "last_updated": "2025-06-20",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2 24B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2-Thinking": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.6
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/Kimi-K2-Thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 2.8
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.6": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 1.9
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 2.56
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Meganova",
    "npm": "@ai-sdk/openai-compatible"
  },
  "merge-gateway": {
    "doc": "https://docs.merge.dev/merge-gateway",
    "env": [
      "MERGE_GATEWAY_API_KEY"
    ],
    "id": "merge-gateway",
    "models": {
      "alibaba/qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "alibaba/qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 250000,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.7-max": {
        "attachment": false,
        "cost": {
          "input": 1.65,
          "output": 4.95
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "alibaba/qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 250000,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4-5-20251001",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-1-20250805",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-5-20251101",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 128000,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-20250514": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-20250514",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-5-20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-a-03-2025": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "cohere/command-a-03-2025",
        "knowledge": "2024-06-01",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/command-r-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere's RAG workhorse for long-context enterprise search and tool use",
        "family": "command-r",
        "id": "cohere/command-r-plus-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R+",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r7b-12-2024": {
        "attachment": false,
        "cost": {
          "input": 0.0375,
          "output": 0.15
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/command-r7b-12-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-12-02",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-02",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "input_audio": 0.3,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview-customtools",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "input_audio": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-flash-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-flash-lite-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-flash-lite-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash-Lite Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Earlier MiniMax agent model for practical coding and productivity tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax/minimax-m2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m3": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 128000,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/codestral-latest": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral code model for completions, refactors, and developer IDE workflows",
        "family": "codestral",
        "id": "mistral/codestral-latest",
        "knowledge": "2024-10",
        "last_updated": "2025-01-04",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-05-29",
        "temperature": true,
        "tool_call": true
      },
      "mistral/devstral-2512": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes",
        "family": "devstral",
        "id": "mistral/devstral-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mistral/devstral-medium-2507": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral/devstral-medium-2507",
        "knowledge": "2025-05",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Medium",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-10",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mistral/devstral-medium-latest": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral/devstral-medium-latest",
        "knowledge": "2025-12",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2 (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mistral/devstral-small-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral/devstral-small-2507",
        "knowledge": "2025-05",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-10",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mistral/magistral-medium-latest": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral-medium",
        "id": "mistral/magistral-medium-latest",
        "knowledge": "2025-06",
        "last_updated": "2025-03-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Medium (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-17",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral/mistral-large-2411",
        "knowledge": "2024-11",
        "last_updated": "2024-11-18",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 2.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-18",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-large-2512": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "mistral/mistral-large-2512",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-large-latest": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral/mistral-large-latest",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-medium-2505": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral/mistral-medium-2505",
        "knowledge": "2025-05",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-medium-latest": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral/mistral-medium-latest",
        "knowledge": "2025-05",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-12",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-small-latest": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral/mistral-small-latest",
        "knowledge": "2025-06",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small (latest)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "mistral/pixtral-large-latest": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral's larger vision model for document-heavy image understanding and chat",
        "family": "pixtral",
        "id": "mistral/pixtral-large-latest",
        "knowledge": "2024-11",
        "last_updated": "2024-11-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code-highspeed": {
        "attachment": true,
        "cost": {
          "input": 1.9,
          "output": 8
        },
        "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code Highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-05-13": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 15
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-05-13",
        "knowledge": "2023-09",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-05-13)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-08-06": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-08-06",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-08-06)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-11-20",
        "knowledge": "2023-09",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5.3-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.15,
                "input": 1.5,
                "output": 9
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt-mini",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4.20-0309-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use",
        "family": "grok",
        "id": "xai/grok-4.20-0309-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "xai/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "zai/glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0,
          "input": 0.2,
          "output": 1.1
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "zai/glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "zai/glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "zai/glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0,
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "zai/glm-4.7-flashx",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-FlashX",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 1,
          "output": 3.2
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "zai/glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "cache_write": 0,
          "input": 1.2,
          "output": 4
        },
        "description": "Faster GLM-5 lane for coding agents that need lower latency",
        "family": "glm",
        "id": "zai/glm-5-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-16",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5.2": {
        "attachment": false,
        "cost": {
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai/glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 50000,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Merge Gateway",
    "npm": "merge-gateway-ai-sdk-provider"
  },
  "minimax": {
    "api": "https://api.minimax.io/anthropic/v1",
    "doc": "https://platform.minimax.io/docs/guides/quickstart",
    "env": [
      "MINIMAX_API_KEY"
    ],
    "id": "minimax",
    "models": {
      "MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows",
        "family": "minimax",
        "id": "MiniMax-M2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Earlier MiniMax agent model for practical coding and productivity tasks",
        "family": "minimax",
        "id": "MiniMax-M2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "MiniMax-M2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "context_over_200k": {
            "cache_read": 0.12,
            "input": 0.6,
            "output": 2.4
          },
          "input": 0.3,
          "output": 1.2,
          "tiers": [
            {
              "cache_read": 0.12,
              "input": 0.6,
              "output": 2.4,
              "tier": {
                "size": 512000,
                "type": "context"
              }
            }
          ]
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "MiniMax-M3",
        "last_updated": "2026-06-25",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "MiniMax (minimax.io)",
    "npm": "@ai-sdk/anthropic"
  },
  "minimax-cn": {
    "api": "https://api.minimaxi.com/anthropic/v1",
    "doc": "https://platform.minimaxi.com/docs/guides/quickstart",
    "env": [
      "MINIMAX_API_KEY"
    ],
    "id": "minimax-cn",
    "models": {
      "MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "context_over_200k": {
            "cache_read": 0.12,
            "input": 0.6,
            "output": 2.4
          },
          "input": 0.3,
          "output": 1.2,
          "tiers": [
            {
              "cache_read": 0.12,
              "input": 0.6,
              "output": 2.4,
              "tier": {
                "size": 512000,
                "type": "context"
              }
            }
          ]
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "MiniMax-M3",
        "last_updated": "2026-06-25",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "MiniMax (minimaxi.com)",
    "npm": "@ai-sdk/anthropic"
  },
  "minimax-cn-coding-plan": {
    "api": "https://api.minimaxi.com/anthropic/v1",
    "doc": "https://platform.minimaxi.com/docs/token-plan/intro",
    "env": [
      "MINIMAX_API_KEY"
    ],
    "id": "minimax-cn-coding-plan",
    "models": {
      "MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M3": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "MiniMax-M3",
        "last_updated": "2026-06-25",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "MiniMax Token Plan (minimaxi.com)",
    "npm": "@ai-sdk/anthropic"
  },
  "minimax-coding-plan": {
    "api": "https://api.minimax.io/anthropic/v1",
    "doc": "https://platform.minimax.io/docs/token-plan/intro",
    "env": [
      "MINIMAX_API_KEY"
    ],
    "id": "minimax-coding-plan",
    "models": {
      "MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "MiniMax-M2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M3": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "MiniMax-M3",
        "last_updated": "2026-06-25",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "MiniMax Token Plan (minimax.io)",
    "npm": "@ai-sdk/anthropic"
  },
  "mistral": {
    "doc": "https://docs.mistral.ai/getting-started/models/",
    "env": [
      "MISTRAL_API_KEY"
    ],
    "id": "mistral",
    "models": {
      "codestral-latest": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral code model for completions, refactors, and developer IDE workflows",
        "family": "codestral",
        "id": "codestral-latest",
        "knowledge": "2024-10",
        "last_updated": "2025-01-04",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-05-29",
        "temperature": true,
        "tool_call": true
      },
      "devstral-2512": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes",
        "family": "devstral",
        "id": "devstral-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "devstral-latest": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "devstral",
        "id": "devstral-latest",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "devstral-medium-2507": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "devstral",
        "id": "devstral-medium-2507",
        "knowledge": "2025-05",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Medium",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-10",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "devstral-medium-latest": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "devstral",
        "id": "devstral-medium-latest",
        "knowledge": "2025-12",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2 (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "devstral-small-2505": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "devstral",
        "id": "devstral-small-2505",
        "knowledge": "2025-05",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small 2505",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-07",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "devstral-small-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "devstral",
        "id": "devstral-small-2507",
        "knowledge": "2025-05",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-10",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "labs-devstral-small-2512": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "devstral",
        "id": "labs-devstral-small-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "magistral-medium-latest": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral-medium",
        "id": "magistral-medium-latest",
        "knowledge": "2025-06",
        "last_updated": "2025-03-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Medium (latest)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-17",
        "temperature": true,
        "tool_call": true
      },
      "magistral-small": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral-small",
        "id": "magistral-small",
        "knowledge": "2025-06",
        "last_updated": "2025-03-17",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Small",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-17",
        "temperature": true,
        "tool_call": true
      },
      "ministral-3b-latest": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3b-latest",
        "knowledge": "2024-10",
        "last_updated": "2024-10-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": true
      },
      "ministral-8b-latest": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-8b-latest",
        "knowledge": "2024-10",
        "last_updated": "2024-10-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 8B (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-embed": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "mistral-embed",
        "id": "mistral-embed",
        "last_updated": "2023-12-11",
        "limit": {
          "context": 8000,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Embed",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-12-11",
        "temperature": false,
        "tool_call": false
      },
      "mistral-large-2411": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-2411",
        "knowledge": "2024-11",
        "last_updated": "2024-11-18",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 2.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-18",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-2512": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "mistral-large-2512",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-large-latest": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-latest",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-2505": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral-medium-2505",
        "knowledge": "2025-05",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-2508": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral-medium-2508",
        "knowledge": "2025-05",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-12",
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-2604": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools",
        "family": "mistral-medium",
        "id": "mistral-medium-2604",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-latest": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral-medium-latest",
        "knowledge": "2025-05",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-12",
        "temperature": true,
        "tool_call": true
      },
      "mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment",
        "family": "mistral-nemo",
        "id": "mistral-nemo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-2506": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-2506",
        "knowledge": "2025-03",
        "last_updated": "2025-06-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-2603": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents",
        "family": "mistral-small",
        "id": "mistral-small-2603",
        "knowledge": "2025-06",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 4",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-latest": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-latest",
        "knowledge": "2025-06",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small (latest)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "open-mistral-7b": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.25
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "open-mistral-7b",
        "knowledge": "2023-12",
        "last_updated": "2023-09-27",
        "limit": {
          "context": 8000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral 7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-09-27",
        "temperature": true,
        "tool_call": true
      },
      "open-mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mistral-nemo",
        "id": "open-mistral-nemo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Open Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "open-mixtral-8x22b": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mixtral",
        "id": "open-mixtral-8x22b",
        "knowledge": "2024-04",
        "last_updated": "2024-04-17",
        "limit": {
          "context": 64000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mixtral 8x22B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-17",
        "temperature": true,
        "tool_call": true
      },
      "open-mixtral-8x7b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mixtral",
        "id": "open-mixtral-8x7b",
        "knowledge": "2024-01",
        "last_updated": "2023-12-11",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mixtral 8x7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-12-11",
        "temperature": true,
        "tool_call": true
      },
      "pixtral-12b": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Mistral vision-language model for image understanding and multimodal chat",
        "family": "pixtral",
        "id": "pixtral-12b",
        "knowledge": "2024-09",
        "last_updated": "2024-09-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-01",
        "temperature": true,
        "tool_call": true
      },
      "pixtral-large-latest": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral's larger vision model for document-heavy image understanding and chat",
        "family": "pixtral",
        "id": "pixtral-large-latest",
        "knowledge": "2024-11",
        "last_updated": "2024-11-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Mistral",
    "npm": "@ai-sdk/mistral"
  },
  "mixlayer": {
    "api": "https://models.mixlayer.ai/v1",
    "doc": "https://docs.mixlayer.com",
    "env": [
      "MIXLAYER_API_KEY"
    ],
    "id": "mixlayer",
    "models": {
      "qwen/qwen3.5-122b-a10b": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 3.2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3.5-122b-a10b",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-27b": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3.5-27b",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-35b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3.5-35b-a3b",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3.5-397b-a17b",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-9b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3.5-9b",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Mixlayer",
    "npm": "@ai-sdk/openai-compatible"
  },
  "moark": {
    "api": "https://moark.com/v1",
    "doc": "https://moark.com/docs/openapi/v1#tag/%E6%96%87%E6%9C%AC%E7%94%9F%E6%88%90",
    "env": [
      "MOARK_API_KEY"
    ],
    "id": "moark",
    "models": {
      "GLM-4.7": {
        "attachment": false,
        "cost": {
          "input": 3.5,
          "output": 14
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "GLM-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "MiniMax-M2.1": {
        "attachment": false,
        "cost": {
          "input": 2.1,
          "output": 8.4
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMax-M2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Moark",
    "npm": "@ai-sdk/openai-compatible"
  },
  "modelscope": {
    "api": "https://api-inference.modelscope.cn/v1",
    "doc": "https://modelscope.cn/docs/model-service/API-Inference/intro",
    "env": [
      "MODELSCOPE_API_KEY"
    ],
    "id": "modelscope",
    "models": {
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-21",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B-Thinking-2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-30",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-30B-A3B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-30B-A3B-Thinking-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-30",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-30",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-30B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "ZhipuAI/GLM-4.5": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "ZhipuAI/GLM-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "ZhipuAI/GLM-4.6": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "ZhipuAI/GLM-4.6",
        "knowledge": "2025-07",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 202752,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "ModelScope",
    "npm": "@ai-sdk/openai-compatible"
  },
  "moonshotai": {
    "api": "https://api.moonshot.ai/v1",
    "doc": "https://platform.moonshot.ai/docs/api/chat",
    "env": [
      "MOONSHOT_API_KEY"
    ],
    "id": "moonshotai",
    "models": {
      "kimi-k2-0711-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0711-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-07-14",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0711",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-14",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-0905-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0905-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.15,
          "output": 8
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking Turbo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-turbo-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.6,
          "input": 2.4,
          "output": 10
        },
        "description": "Fast Kimi model for responsive chat, coding help, and agent loops",
        "family": "kimi-k2",
        "id": "kimi-k2-turbo-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.7-code-highspeed": {
        "attachment": true,
        "cost": {
          "cache_read": 0.38,
          "input": 1.9,
          "output": 8
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code HighSpeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Moonshot AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "moonshotai-cn": {
    "api": "https://api.moonshot.cn/v1",
    "doc": "https://platform.moonshot.cn/docs/api/chat",
    "env": [
      "MOONSHOT_API_KEY"
    ],
    "id": "moonshotai-cn",
    "models": {
      "kimi-k2-0711-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0711-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-07-14",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0711",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-14",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-0905-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "kimi-k2-0905-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.15,
          "output": 8
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking Turbo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-turbo-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.6,
          "input": 2.4,
          "output": 10
        },
        "description": "Fast Kimi model for responsive chat, coding help, and agent loops",
        "family": "kimi-k2",
        "id": "kimi-k2-turbo-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2.7-code-highspeed": {
        "attachment": true,
        "cost": {
          "cache_read": 0.38,
          "input": 1.9,
          "output": 8
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code HighSpeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Moonshot AI (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "morph": {
    "api": "https://api.morphllm.com/v1",
    "doc": "https://docs.morphllm.com/api-reference/introduction",
    "env": [
      "MORPH_API_KEY"
    ],
    "id": "morph",
    "models": {
      "auto": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 1.55
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "auto",
        "id": "auto",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "temperature": false,
        "tool_call": false
      },
      "morph-v3-fast": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "morph",
        "id": "morph-v3-fast",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 16000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph v3 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": false,
        "tool_call": false
      },
      "morph-v3-large": {
        "attachment": false,
        "cost": {
          "input": 0.9,
          "output": 1.9
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "morph",
        "id": "morph-v3-large",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph v3 Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "Morph",
    "npm": "@ai-sdk/openai-compatible"
  },
  "nano-gpt": {
    "api": "https://nano-gpt.com/api/v1",
    "doc": "https://docs.nano-gpt.com",
    "env": [
      "NANO_GPT_API_KEY"
    ],
    "id": "nano-gpt",
    "models": {
      "Alibaba-NLP/Tongyi-DeepResearch-30B-A3B": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.24000000000000002
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "yi",
        "id": "Alibaba-NLP/Tongyi-DeepResearch-30B-A3B",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tongyi DeepResearch 30B A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-26",
        "structured_output": false,
        "tool_call": false
      },
      "Baichuan-M2": {
        "attachment": false,
        "cost": {
          "input": 15.73,
          "output": 15.73
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Baichuan-M2",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baichuan M2 32B Medical",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-19",
        "structured_output": false,
        "tool_call": false
      },
      "Baichuan4-Air": {
        "attachment": false,
        "cost": {
          "input": 0.157,
          "output": 0.157
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Baichuan4-Air",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baichuan 4 Air",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-19",
        "structured_output": false,
        "tool_call": false
      },
      "Baichuan4-Turbo": {
        "attachment": false,
        "cost": {
          "input": 2.42,
          "output": 2.42
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Baichuan4-Turbo",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Baichuan 4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-19",
        "structured_output": false,
        "tool_call": false
      },
      "Doctor-Shotgun/MS3.2-24B-Magnum-Diamond": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "Doctor-Shotgun/MS3.2-24B-Magnum-Diamond",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MS3.2 24B Magnum Diamond",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-24",
        "structured_output": false,
        "tool_call": false
      },
      "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0": {
        "attachment": false,
        "cost": {
          "input": 2.006,
          "output": 2.006
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "EVA Llama 3.33 70B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1": {
        "attachment": false,
        "cost": {
          "input": 2.006,
          "output": 2.006
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "EVA-LLaMA-3.33-70B-v0.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2": {
        "attachment": false,
        "cost": {
          "input": 0.7989999999999999,
          "output": 0.7989999999999999
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "EVA-Qwen2.5-32B-v0.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2": {
        "attachment": false,
        "cost": {
          "input": 0.7989999999999999,
          "output": 0.7989999999999999
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "EVA-Qwen2.5-72B-v0.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "llama",
        "id": "Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.05 Storybreaker Ministral 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": false,
        "tool_call": false
      },
      "Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Tenyxchat Storybreaker 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": false,
        "tool_call": false
      },
      "GLM-4.6-Derestricted-v5": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "GLM-4.6-Derestricted-v5",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6 Derestricted v5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-23",
        "structured_output": false,
        "tool_call": false
      },
      "GalrionSoftworks/MN-LooseCannon-12B-v1": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "GalrionSoftworks/MN-LooseCannon-12B-v1",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MN-LooseCannon-12B-v1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0306,
          "input": 0.306,
          "output": 0.306
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Claude 4.6 Opus Reasoning Distilled",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-Cognitive-Unshackled": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-Cognitive-Unshackled",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Cognitive Unshackled",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-DarkIdol": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-DarkIdol",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B DarkIdol",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-GarnetV2": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-GarnetV2",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Garnet V2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-Gemopus": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-Gemopus",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Gemopus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-Musica-v1": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-Musica-v1",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Musica v1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-Queen": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-Queen",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Queen",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "Gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Gemma-4-31B-it",
        "last_updated": "2026-04-09",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-09",
        "structured_output": false,
        "tool_call": false
      },
      "Gryphe/MythoMax-L2-13b": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1003
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Gryphe/MythoMax-L2-13b",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 4000,
          "input": 4000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MythoMax 13B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-08",
        "structured_output": false,
        "tool_call": false
      },
      "Infermatic/MN-12B-Inferor-v0.0": {
        "attachment": false,
        "cost": {
          "input": 0.25499999999999995,
          "output": 0.49299999999999994
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "Infermatic/MN-12B-Inferor-v0.0",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo Inferor 12B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "KAT-Coder-Air-V1": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "KAT-Coder-Air-V1",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KAT Coder Air V1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-28",
        "structured_output": false,
        "tool_call": false
      },
      "KAT-Coder-Exp-72B-1010": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "KAT-Coder-Exp-72B-1010",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KAT Coder Exp 72B 1010",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-28",
        "structured_output": false,
        "tool_call": false
      },
      "LLM360/K2-Think": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "LLM360/K2-Think",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "K2-Think",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "LatitudeGames/Wayfarer-Large-70B-Llama-3.3": {
        "attachment": false,
        "cost": {
          "input": 0.700000007,
          "output": 0.700000007
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "LatitudeGames/Wayfarer-Large-70B-Llama-3.3",
        "last_updated": "2025-02-20",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Wayfarer",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-20",
        "structured_output": false,
        "tool_call": false
      },
      "Magistral-Small-2506": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Magistral-Small-2506",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Small 2506",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "MarinaraSpaghetti/NemoMix-Unleashed-12B": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "MarinaraSpaghetti/NemoMix-Unleashed-12B",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NemoMix 12B Unleashed",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "Meta-Llama-3-1-8B-Instruct-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.03
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Meta-Llama-3-1-8B-Instruct-FP8",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B (decentralized)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": false,
        "tool_call": false
      },
      "MiniMax-M1": {
        "attachment": false,
        "cost": {
          "input": 0.1394,
          "output": 1.3328
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "MiniMax-M1",
        "last_updated": "2025-06-16",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-16",
        "structured_output": false,
        "tool_call": false
      },
      "MiniMax-M2": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 1.53
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "MiniMax-M2",
        "last_updated": "2025-10-25",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-25",
        "structured_output": false,
        "tool_call": false
      },
      "MiniMaxAI/MiniMax-M1-80k": {
        "attachment": false,
        "cost": {
          "input": 0.6052,
          "output": 2.4225000000000003
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M1-80k",
        "last_updated": "2025-06-16",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M1 80K",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-16",
        "structured_output": false,
        "tool_call": false
      },
      "NeverSleep/Lumimaid-v0.2-70B": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1.5
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "NeverSleep/Lumimaid-v0.2-70B",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Lumimaid v0.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "NousResearch/DeepHermes-3-Mistral-24B-Preview": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "NousResearch/DeepHermes-3-Mistral-24B-Preview",
        "last_updated": "2025-05-10",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepHermes-3 Mistral 24B (Preview)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-10",
        "structured_output": false,
        "tool_call": false
      },
      "NousResearch/Hermes-4-70B:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.3995
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "NousResearch/Hermes-4-70B:thinking",
        "last_updated": "2025-09-17",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 (Thinking)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-17",
        "structured_output": false,
        "tool_call": false
      },
      "NousResearch/hermes-3-llama-3.1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.408,
          "output": 0.408
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "NousResearch/hermes-3-llama-3.1-70b",
        "last_updated": "2026-01-07",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 3 70B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-07",
        "structured_output": false,
        "tool_call": false
      },
      "NousResearch/hermes-4-405b": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "NousResearch/hermes-4-405b",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-26",
        "structured_output": true,
        "tool_call": false
      },
      "NousResearch/hermes-4-405b:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "NousResearch/hermes-4-405b:thinking",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 Large (Thinking)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": true,
        "tool_call": false
      },
      "NousResearch/hermes-4-70b": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.3995
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "NousResearch/hermes-4-70b",
        "last_updated": "2025-07-03",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 Medium",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-03",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen2.5-32B-EVA-v0.2": {
        "attachment": false,
        "cost": {
          "input": 0.493,
          "output": 0.493
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen2.5-32B-EVA-v0.2",
        "last_updated": "2024-09-01",
        "limit": {
          "context": 24576,
          "input": 24576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 32b EVA",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-01",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Anko": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Anko",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Anko",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-BlueStar-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-BlueStar-Derestricted",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B BlueStar Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-BlueStar-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-BlueStar-Derestricted-Lite",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B BlueStar Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-BlueStar-v2-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-BlueStar-v2-Derestricted",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B BlueStar v2 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-BlueStar-v2-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-BlueStar-v2-Derestricted-Lite",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B BlueStar v2 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-BlueStar-v3-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-BlueStar-v3-Derestricted",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B BlueStar v3 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-BlueStar-v3-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-BlueStar-v3-Derestricted-Lite",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B BlueStar v3 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Derestricted",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Infracelestial": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Infracelestial",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Infracelestial",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Marvin-DPO-V2-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Marvin-DPO-V2-Derestricted",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Marvin DPO V2 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Marvin-DPO-V2-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Marvin-DPO-V2-Derestricted-Lite",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Marvin DPO V2 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Marvin-V2-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Marvin-V2-Derestricted",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Marvin V2 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Marvin-V2-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Marvin-V2-Derestricted-Lite",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Marvin V2 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Musica-v1": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Musica-v1",
        "last_updated": "2026-03-27",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Musica v1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-NaNovel-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-NaNovel-Derestricted",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B NaNovel Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-NaNovel-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-NaNovel-Derestricted-Lite",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B NaNovel Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Omega Evolution v2.0 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Omega-Evolution-v2.0-Derestricted-Lite",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Omega Evolution v2.0 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Omega Evolution v2.2 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Omega-Evolution-v2.2-Derestricted-Lite",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Omega Evolution v2.2 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Queen-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Queen-Derestricted",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Queen Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Queen-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Queen-Derestricted-Lite",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Queen Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-RpRMax-v1": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-RpRMax-v1",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B RpRMax v1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Vivid-Durian": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Vivid-Durian",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Vivid Durian",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-18",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Writer-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Writer-Derestricted",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Writer Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Writer-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Writer-Derestricted-Lite",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Writer Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Writer-V2-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Writer-V2-Derestricted",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Writer V2 Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-Writer-V2-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-Writer-V2-Derestricted-Lite",
        "last_updated": "2026-04-06",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Writer V2 Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-06",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-earica-Derestricted": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-earica-Derestricted",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B earica Derestricted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "Qwen3.5-27B-earica-Derestricted-Lite": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "Qwen3.5-27B-earica-Derestricted-Lite",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B earica Derestricted Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 0.5
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Omega Directive 24B Unslop v2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-08",
        "structured_output": false,
        "tool_call": false
      },
      "Salesforce/Llama-xLAM-2-70b-fc-r": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 2.5
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Salesforce/Llama-xLAM-2-70b-fc-r",
        "last_updated": "2025-04-13",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-xLAM-2 70B fc-r",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-13",
        "structured_output": false,
        "tool_call": false
      },
      "Sao10K/L3-8B-Stheno-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Sao10K/L3-8B-Stheno-v3.2",
        "last_updated": "2024-11-29",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10K Stheno 8b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-29",
        "structured_output": false,
        "tool_call": false
      },
      "Sao10K/L3.1-70B-Euryale-v2.2": {
        "attachment": false,
        "cost": {
          "input": 0.306,
          "output": 0.357
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Sao10K/L3.1-70B-Euryale-v2.2",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 20480,
          "input": 20480,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Euryale",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": false,
        "tool_call": false
      },
      "Sao10K/L3.1-70B-Hanami-x1": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Sao10K/L3.1-70B-Hanami-x1",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Hanami",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": false,
        "tool_call": false
      },
      "Sao10K/L3.3-70B-Euryale-v2.3": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Sao10K/L3.3-70B-Euryale-v2.3",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 20480,
          "input": 20480,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Euryale",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "Steelskull/L3.3-Cu-Mai-R1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Steelskull/L3.3-Cu-Mai-R1-70b",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Cu Mai",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "Steelskull/L3.3-Electra-R1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.69989,
          "output": 0.69989
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Steelskull/L3.3-Electra-R1-70b",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Steelskull Electra R1 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "Steelskull/L3.3-MS-Evalebis-70b": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Steelskull/L3.3-MS-Evalebis-70b",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MS Evalebis 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "Steelskull/L3.3-MS-Evayale-70B": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Steelskull/L3.3-MS-Evayale-70B",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Evayale 70b ",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "Steelskull/L3.3-MS-Nevoria-70b": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Steelskull/L3.3-MS-Nevoria-70b",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Steelskull Nevoria 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "Steelskull/L3.3-Nevoria-R1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "Steelskull/L3.3-Nevoria-R1-70b",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Steelskull Nevoria R1 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 2.5
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "TEE/deepseek-v3.1",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 164000,
          "input": 164000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-21",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "TEE/deepseek-v3.2",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 164000,
          "input": 164000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 5.25
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "TEE/deepseek-v4-pro",
        "last_updated": "2026-04-25",
        "limit": {
          "context": 800000,
          "input": 800000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-25",
        "structured_output": true,
        "tool_call": true
      },
      "TEE/deepseek-v4-pro:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 5.25
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "TEE/deepseek-v4-pro:thinking",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 800000,
          "input": 800000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro Thinking TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "TEE/gemma-3-27b-it": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "TEE/gemma-3-27b-it",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 27B TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-10",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/gemma-4-26b-a4b-uncensored": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.7
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "TEE/gemma-4-26b-a4b-uncensored",
        "last_updated": "2026-05-23",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B Uncensored TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-23",
        "structured_output": true,
        "tool_call": true
      },
      "TEE/gemma-4-31b-it": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.46
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "TEE/gemma-4-31b-it",
        "last_updated": "2026-05-26",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-26",
        "structured_output": false,
        "tool_call": true
      },
      "TEE/gemma4-31b": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 1
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "TEE/gemma4-31b",
        "last_updated": "2026-04-04",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-04",
        "structured_output": true,
        "tool_call": false
      },
      "TEE/gemma4-31b:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 1
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "TEE/gemma4-31b:thinking",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Thinking TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-02",
        "structured_output": true,
        "tool_call": false
      },
      "TEE/glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 3.3
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "TEE/glm-4.7",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 131000,
          "input": 131000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-29",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.5
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "TEE/glm-4.7-flash",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 203000,
          "input": 203000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-19",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/glm-5": {
        "attachment": false,
        "cost": {
          "input": 1.2,
          "output": 3.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "TEE/glm-5",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 203000,
          "input": 203000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-11",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 1.5,
          "output": 5.25
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "TEE/glm-5.1",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 202752,
          "input": 202752,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-20",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/glm-5.1-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 1.5,
          "output": 5.25
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "TEE/glm-5.1-thinking",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 202752,
          "input": 202752,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1 Thinking TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-20",
        "structured_output": true,
        "tool_call": true
      },
      "TEE/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "TEE/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 120B TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "TEE/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 20B TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/kimi-k2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.9
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "TEE/kimi-k2.5",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-29",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/kimi-k2.5-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.9
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "TEE/kimi-k2.5-thinking",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 Thinking TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.375,
          "input": 1.5,
          "output": 5.25
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "TEE/kimi-k2.6",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "structured_output": false,
        "tool_call": true
      },
      "TEE/llama3-3-70b": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 2
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "TEE/llama3-3-70b",
        "last_updated": "2025-07-03",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-03",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.38
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "TEE/minimax-m2.5",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 196608,
          "input": 196608,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5 TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-20",
        "structured_output": true,
        "tool_call": true
      },
      "TEE/qwen2.5-vl-72b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "TEE/qwen2.5-vl-72b-instruct",
        "last_updated": "2025-02-01",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 VL 72B TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-01",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/qwen3-30b-a3b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.44999999999999996
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "TEE/qwen3-30b-a3b-instruct-2507",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 262000,
          "input": 262000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct 2507 TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-29",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/qwen3.5-122b-a10b": {
        "attachment": false,
        "cost": {
          "input": 0.46,
          "output": 3.68
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "TEE/qwen3.5-122b-a10b",
        "last_updated": "2026-05-26",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B A10B TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-26",
        "structured_output": false,
        "tool_call": true
      },
      "TEE/qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.4
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "TEE/qwen3.5-27b",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-13",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/qwen3.5-397b-a17b": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "TEE/qwen3.5-397b-a17b",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 258048,
          "input": 258048,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B TEE",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-28",
        "structured_output": false,
        "tool_call": false
      },
      "TEE/qwen3.6-35b-a3b-uncensored": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "TEE/qwen3.6-35b-a3b-uncensored",
        "last_updated": "2026-05-23",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B Uncensored TEE",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-23",
        "structured_output": true,
        "tool_call": true
      },
      "THUDM/GLM-4-32B-0414": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "THUDM/GLM-4-32B-0414",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4 32B 0414",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": false,
        "tool_call": false
      },
      "THUDM/GLM-4-9B-0414": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "THUDM/GLM-4-9B-0414",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4 9B 0414",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": false,
        "tool_call": false
      },
      "THUDM/GLM-Z1-32B-0414": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm-z",
        "id": "THUDM/GLM-Z1-32B-0414",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM Z1 32B 0414",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "THUDM/GLM-Z1-9B-0414": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm-z",
        "id": "THUDM/GLM-Z1-9B-0414",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM Z1 9B 0414",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Anubis-70B-v1": {
        "attachment": false,
        "cost": {
          "input": 0.31,
          "output": 0.31
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Anubis-70B-v1",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anubis 70B v1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Anubis-70B-v1.1": {
        "attachment": false,
        "cost": {
          "input": 0.31,
          "output": 0.31
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Anubis-70B-v1.1",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anubis 70B v1.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Cydonia-24B-v2": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1207
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Cydonia-24B-v2",
        "last_updated": "2025-02-17",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "The Drummer Cydonia 24B v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-17",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Cydonia-24B-v4": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2414
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Cydonia-24B-v4",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "The Drummer Cydonia 24B v4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-22",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Cydonia-24B-v4.1": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1207
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Cydonia-24B-v4.1",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "The Drummer Cydonia 24B v4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-19",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Cydonia-24B-v4.3": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1207
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Cydonia-24B-v4.3",
        "last_updated": "2025-12-25",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "The Drummer Cydonia 24B v4.3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-25",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Magidonia-24B-v4.3": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1207
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Magidonia-24B-v4.3",
        "last_updated": "2025-12-25",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "The Drummer Magidonia 24B v4.3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-25",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Rocinante-12B-v1.1": {
        "attachment": false,
        "cost": {
          "input": 0.408,
          "output": 0.595
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "TheDrummer/Rocinante-12B-v1.1",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Rocinante 12b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/Skyfall-31B-v4.2": {
        "attachment": true,
        "cost": {
          "input": 0.55,
          "output": 0.8
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "TheDrummer/Skyfall-31B-v4.2",
        "last_updated": "2026-03-26",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TheDrummer Skyfall 31B v4.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-26",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/UnslopNemo-12B-v4.1": {
        "attachment": true,
        "cost": {
          "input": 0.493,
          "output": 0.493
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "TheDrummer/UnslopNemo-12B-v4.1",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "UnslopNemo 12b v4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "TheDrummer/skyfall-36b-v2": {
        "attachment": true,
        "cost": {
          "input": 0.493,
          "output": 0.493
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "TheDrummer/skyfall-36b-v2",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TheDrummer Skyfall 36B V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-10",
        "structured_output": false,
        "tool_call": false
      },
      "Tongyi-Zhiwen/QwenLong-L1-32B": {
        "attachment": false,
        "cost": {
          "input": 0.13999999999999999,
          "output": 0.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Tongyi-Zhiwen/QwenLong-L1-32B",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "QwenLong L1 32B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-25",
        "structured_output": false,
        "tool_call": false
      },
      "Unbabel/M-Prometheus-14B": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "Unbabel/M-Prometheus-14B",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "M-Prometheus 14B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-29",
        "structured_output": false,
        "tool_call": false
      },
      "VongolaChouko/Starcannon-Unleashed-12B-v1.0": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "VongolaChouko/Starcannon-Unleashed-12B-v1.0",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo Starcannon 12b v1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "abacusai/Dracarys-72B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "abacusai/Dracarys-72B-Instruct",
        "last_updated": "2025-08-02",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Dracarys 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-02",
        "structured_output": false,
        "tool_call": false
      },
      "aion-labs/aion-1.0": {
        "attachment": false,
        "cost": {
          "input": 3.995,
          "output": 7.99
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "aion-labs/aion-1.0",
        "last_updated": "2025-02-01",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-01",
        "structured_output": false,
        "tool_call": false
      },
      "aion-labs/aion-1.0-mini": {
        "attachment": false,
        "cost": {
          "input": 0.7989999999999999,
          "output": 1.394
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek",
        "id": "aion-labs/aion-1.0-mini",
        "last_updated": "2025-02-20",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion 1.0 mini (DeepSeek)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-20",
        "structured_output": false,
        "tool_call": false
      },
      "aion-labs/aion-2.0": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.6
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "aion-labs/aion-2.0",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AionLabs: Aion-2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-23",
        "structured_output": false,
        "tool_call": false
      },
      "aion-labs/aion-2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.35,
          "input": 1,
          "output": 3
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "aion-labs/aion-2.5",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AionLabs: Aion-2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-20",
        "structured_output": false,
        "tool_call": false
      },
      "aion-labs/aion-rp-llama-3.1-8b": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "aion-labs/aion-rp-llama-3.1-8b",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8b (uncensored)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": false,
        "tool_call": false
      },
      "alibaba/qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.203,
          "output": 2.24
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "alibaba/qwen3.6-27b",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-23",
        "structured_output": false,
        "tool_call": false
      },
      "alibaba/qwen3.6-27b:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.203,
          "output": 2.24
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "alibaba/qwen3.6-27b:thinking",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 131072,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": false,
        "tool_call": false
      },
      "alibaba/qwen3.6-flash": {
        "attachment": false,
        "cost": {
          "input": 0.19,
          "output": 1.16
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "alibaba/qwen3.6-flash",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 991800,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-17",
        "structured_output": false,
        "tool_call": false
      },
      "allenai/olmo-3-32b-think": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.44999999999999996
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "allenai",
        "id": "allenai/olmo-3-32b-think",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Olmo 3 32B Think",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-01",
        "structured_output": false,
        "tool_call": false
      },
      "amazon/nova-2-lite-v1": {
        "attachment": false,
        "cost": {
          "input": 0.5099999999999999,
          "output": 4.25
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova",
        "id": "amazon/nova-2-lite-v1",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon Nova 2 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "structured_output": false,
        "tool_call": false
      },
      "amazon/nova-lite-v1": {
        "attachment": false,
        "cost": {
          "input": 0.0595,
          "output": 0.238
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-lite",
        "id": "amazon/nova-lite-v1",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "input": 300000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon Nova Lite 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "structured_output": false,
        "tool_call": false
      },
      "amazon/nova-micro-v1": {
        "attachment": false,
        "cost": {
          "input": 0.0357,
          "output": 0.1394
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-micro",
        "id": "amazon/nova-micro-v1",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon Nova Micro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "structured_output": false,
        "tool_call": false
      },
      "amazon/nova-pro-v1": {
        "attachment": false,
        "cost": {
          "input": 0.7989999999999999,
          "output": 3.1959999999999997
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova-pro",
        "id": "amazon/nova-pro-v1",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "input": 300000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amazon Nova Pro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "structured_output": false,
        "tool_call": false
      },
      "anthracite-org/magnum-v2-72b": {
        "attachment": false,
        "cost": {
          "input": 2.006,
          "output": 2.992
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "anthracite-org/magnum-v2-72b",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magnum V2 72B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "anthracite-org/magnum-v4-72b": {
        "attachment": true,
        "cost": {
          "input": 2.006,
          "output": 2.992
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "anthracite-org/magnum-v4-72b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magnum v4 72B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "anthropic/claude-haiku-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "anthropic/claude-haiku-latest",
        "last_updated": "2026-03-29",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-29",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.6 Opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6:thinking": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6:thinking",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.6 Opus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6:thinking:low": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6:thinking:low",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.6 Opus Thinking Low",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6:thinking:max": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6:thinking:max",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.6 Opus Thinking Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6:thinking:medium": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6:thinking:medium",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.6 Opus Thinking Medium",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4998,
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.7",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.7 Opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4998,
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.7:thinking",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.7 Opus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4998,
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4998,
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.8:thinking",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-28",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-opus-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4998,
          "input": 4.998,
          "output": 25.007
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-latest",
        "last_updated": "2026-03-29",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-29",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.993999999999998
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-17",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6:thinking": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.993999999999998
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6:thinking",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-17",
        "structured_output": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2992,
          "input": 2.992,
          "output": 14.994
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-latest",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-01",
        "structured_output": true,
        "tool_call": true
      },
      "arcee-ai/trinity-large-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.9
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "arcee-ai/trinity-large-thinking",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Large Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "structured_output": false,
        "tool_call": true
      },
      "arcee-ai/trinity-mini": {
        "attachment": false,
        "cost": {
          "input": 0.045000000000000005,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "trinity-mini",
        "id": "arcee-ai/trinity-mini",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "structured_output": false,
        "tool_call": false
      },
      "asi1-mini": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "asi1-mini",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ASI1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-25",
        "structured_output": false,
        "tool_call": false
      },
      "auto-model": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "auto-model",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto model",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "auto-model-basic": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "auto-model-basic",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto model (Basic)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "auto-model-premium": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "auto-model-premium",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto model (Premium)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "auto-model-standard": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "auto-model-standard",
        "last_updated": "2024-06-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto model (Standard)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "azure-gpt-4-turbo": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 30.005
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "azure-gpt-4-turbo",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Azure gpt-4-turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "tool_call": false
      },
      "azure-gpt-4o": {
        "attachment": true,
        "cost": {
          "input": 2.499,
          "output": 9.996
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "azure-gpt-4o",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Azure gpt-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "tool_call": true
      },
      "azure-gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "input": 0.1496,
          "output": 0.595
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "azure-gpt-4o-mini",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Azure gpt-4o-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "tool_call": true
      },
      "azure-o1": {
        "attachment": false,
        "cost": {
          "input": 14.994,
          "output": 59.993
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "azure-o1",
        "last_updated": "2024-12-17",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Azure o1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "azure-o3-mini": {
        "attachment": false,
        "cost": {
          "input": 1.088,
          "output": 4.3996
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "azure-o3-mini",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Azure o3-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-31",
        "structured_output": false,
        "tool_call": false
      },
      "baidu/ernie-4.5-vl-28b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.13999999999999999,
          "output": 0.5599999999999999
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "ernie",
        "id": "baidu/ernie-4.5-vl-28b-a3b",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 VL 28B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-30",
        "structured_output": false,
        "tool_call": false
      },
      "baseten/Kimi-K2-Instruct-FP4": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "baseten/Kimi-K2-Instruct-FP4",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0711 Instruct FP4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-11",
        "structured_output": false,
        "tool_call": false
      },
      "brave": {
        "attachment": false,
        "cost": {
          "input": 5,
          "output": 5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "brave",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Brave (Answers)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-02",
        "structured_output": false,
        "tool_call": false
      },
      "brave-pro": {
        "attachment": false,
        "cost": {
          "input": 5,
          "output": 5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "brave-pro",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Brave (Pro)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-02",
        "structured_output": false,
        "tool_call": false
      },
      "brave-research": {
        "attachment": false,
        "cost": {
          "input": 5,
          "output": 5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "brave-research",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Brave (Research)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-02",
        "structured_output": false,
        "tool_call": false
      },
      "bytedance-seed/seed-2.0-lite": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "bytedance-seed/seed-2.0-lite",
        "last_updated": "2026-03-10",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance Seed 2.0 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-10",
        "structured_output": true,
        "tool_call": false
      },
      "chutesai/Mistral-Small-3.2-24B-Instruct-2506": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.4
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "chutesai",
        "id": "chutesai/Mistral-Small-3.2-24B-Instruct-2506",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2 24b Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "claude-3-5-haiku-20241022": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-3-5-haiku-20241022",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-haiku-4-5-20251001",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "structured_output": true,
        "tool_call": true
      },
      "claude-haiku-4-5-20251001-thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 1,
          "output": 5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-haiku-4-5-20251001-thinking",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-1-20250805": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-1-20250805",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-1-thinking": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-1-thinking",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-1-thinking:1024": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-1-thinking:1024",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus Thinking (1K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-1-thinking:32000": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-1-thinking:32000",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus Thinking (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-1-thinking:32768": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-1-thinking:32768",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus Thinking (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-1-thinking:8192": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-1-thinking:8192",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus Thinking (8K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-20250514",
        "last_updated": "2025-05-14",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Opus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-14",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-5-20251101",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.5 Opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-01",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101:thinking": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-5-20251101:thinking",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.5 Opus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-01",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-thinking": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-thinking",
        "last_updated": "2025-07-15",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Opus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-15",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-thinking:1024": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-thinking:1024",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Opus Thinking (1K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-thinking:32000": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-thinking:32000",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Opus Thinking (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-thinking:32768": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-thinking:32768",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Opus Thinking (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-opus-4-thinking:8192": {
        "attachment": true,
        "cost": {
          "input": 14.994,
          "output": 75.004
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-opus-4-thinking:8192",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Opus Thinking (8K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-20250514": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-20250514",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Sonnet",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-29",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-5-20250929",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-29",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929-thinking": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-5-20250929-thinking",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-thinking": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-thinking",
        "last_updated": "2025-02-24",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Sonnet Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-24",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-thinking:1024": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-thinking:1024",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Sonnet Thinking (1K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-thinking:32768": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-thinking:32768",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Sonnet Thinking (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-thinking:64000": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-thinking:64000",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Sonnet Thinking (64K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claude-sonnet-4-thinking:8192": {
        "attachment": true,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claude-sonnet-4-thinking:8192",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4 Sonnet Thinking (8K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "structured_output": true,
        "tool_call": true
      },
      "claw-high": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claw-high",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claw High",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": true,
        "tool_call": true
      },
      "claw-low": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claw-low",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claw Low",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": true,
        "tool_call": true
      },
      "claw-medium": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "claw-medium",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 204800,
          "input": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claw Medium",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": true,
        "tool_call": true
      },
      "cognitivecomputations/dolphin-2.9.2-qwen2-72b": {
        "attachment": false,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "cognitivecomputations/dolphin-2.9.2-qwen2-72b",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Dolphin 72b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": false,
        "tool_call": false
      },
      "cohere/command-r": {
        "attachment": false,
        "cost": {
          "input": 0.476,
          "output": 1.428
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/command-r",
        "last_updated": "2024-03-11",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere: Command R",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-11",
        "structured_output": false,
        "tool_call": false
      },
      "cohere/command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.856,
          "output": 14.246
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/command-r-plus-08-2024",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere: Command R+",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-30",
        "structured_output": false,
        "tool_call": true
      },
      "command-a-plus-05-2026": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "command-a-plus-05-2026",
        "last_updated": "2026-05-22",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command A+ (05/2026)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-22",
        "structured_output": true,
        "tool_call": false
      },
      "command-a-reasoning-08-2025": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "command-a-reasoning-08-2025",
        "last_updated": "2025-08-22",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Command A (08/2025)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-22",
        "structured_output": false,
        "tool_call": false
      },
      "deepclaude": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "deepclaude",
        "last_updated": "2025-02-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepClaude",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-01",
        "structured_output": false,
        "tool_call": false
      },
      "deepcogito/cogito-v1-preview-qwen-32B": {
        "attachment": false,
        "cost": {
          "input": 1.7999999999999998,
          "output": 1.7999999999999998
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "deepcogito/cogito-v1-preview-qwen-32B",
        "last_updated": "2025-05-10",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cogito v1 Preview Qwen 32B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-10",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-R1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.7
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-R1-0528",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-V3.1": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.7
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-V3.1-Terminus": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.7
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1-Terminus",
        "last_updated": "2025-08-02",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Terminus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-02",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1-Terminus:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.7
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V3.1-Terminus:thinking",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Terminus (Thinking)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-22",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.7
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V3.1:thinking",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-21",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-ai/deepseek-v3.2-exp": {
        "attachment": false,
        "cost": {
          "input": 0.27999999999999997,
          "output": 0.42000000000000004
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/deepseek-v3.2-exp",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163840,
          "input": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Exp",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-29",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-ai/deepseek-v3.2-exp-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.27999999999999997,
          "output": 0.42000000000000004
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/deepseek-v3.2-exp-thinking",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163840,
          "input": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Exp Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-chat": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 0.7
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "deepseek-chat",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3/Deepseek Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek-chat-cheaper": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 0.7
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "deepseek-chat-cheaper",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3/Chat Cheaper",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek-math-v2": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "deepseek-math-v2",
        "last_updated": "2025-12-03",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Math V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-03",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.7
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "deepseek-r1",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-r1-sambanova": {
        "attachment": false,
        "cost": {
          "input": 4.998,
          "output": 6.987
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "deepseek-r1-sambanova",
        "last_updated": "2025-02-20",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-20",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-reasoner": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.7
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "deepseek-reasoner",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Reasoner",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-20",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-reasoner-cheaper": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.7
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "deepseek-reasoner-cheaper",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek R1 Cheaper",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-20",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.7
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "deepseek-v3-0324",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Chat 0324",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-24",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-latest": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 2.2
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-latest",
        "last_updated": "2026-05-03",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-03",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-prover-v2-671b": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 2.5
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek",
        "id": "deepseek/deepseek-prover-v2-671b",
        "last_updated": "2025-04-30",
        "limit": {
          "context": 160000,
          "input": 160000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Prover v2 671B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-30",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek/deepseek-v3.2": {
        "attachment": true,
        "cost": {
          "input": 0.27999999999999997,
          "output": 0.42000000000000004
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163000,
          "input": 163000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-speciale": {
        "attachment": true,
        "cost": {
          "input": 0.27999999999999997,
          "output": 0.42000000000000004
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2-speciale",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 163000,
          "input": 163000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Speciale",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-02",
        "structured_output": false,
        "tool_call": false
      },
      "deepseek/deepseek-v3.2:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.27999999999999997,
          "output": 0.42000000000000004
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2:thinking",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163000,
          "input": 163000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "id": "deepseek/deepseek-v4-flash",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "id": "deepseek/deepseek-v4-flash:thinking",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 2.2
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "deepseek/deepseek-v4-pro",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro-cheaper": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "deepseek/deepseek-v4-pro-cheaper",
        "last_updated": "2026-04-25",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro Cheaper",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-25",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro-cheaper:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "deepseek/deepseek-v4-pro-cheaper:thinking",
        "last_updated": "2026-04-25",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro Cheaper (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-25",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 2.2
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "deepseek/deepseek-v4-pro:thinking",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "tool_call": true
      },
      "dmind/dmind-1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.6
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "dmind/dmind-1",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DMind-1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "dmind/dmind-1-mini": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "dmind/dmind-1-mini",
        "last_updated": "2025-06-01",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DMind-1-Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-1-5-thinking-pro-250415": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-1-5-thinking-pro-250415",
        "last_updated": "2025-04-17",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Thinking Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-17",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-1-5-thinking-pro-vision-250415": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-1-5-thinking-pro-vision-250415",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Thinking Pro Vision",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-1-5-thinking-vision-pro-250428": {
        "attachment": true,
        "cost": {
          "input": 0.55,
          "output": 1.43
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-1-5-thinking-vision-pro-250428",
        "last_updated": "2025-05-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Thinking Vision Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-15",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-1.5-pro-256k": {
        "attachment": false,
        "cost": {
          "input": 0.799,
          "output": 1.445
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-1.5-pro-256k",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Pro 256k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-1.5-pro-32k": {
        "attachment": false,
        "cost": {
          "input": 0.1343,
          "output": 0.3349
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-1.5-pro-32k",
        "last_updated": "2025-01-22",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Pro 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-22",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-1.5-vision-pro-32k": {
        "attachment": true,
        "cost": {
          "input": 0.459,
          "output": 1.377
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-1.5-vision-pro-32k",
        "last_updated": "2025-01-22",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Vision Pro 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-22",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-1-6-250615": {
        "attachment": false,
        "cost": {
          "input": 0.204,
          "output": 0.51
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-1-6-250615",
        "last_updated": "2025-06-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 1.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-15",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-1-6-flash-250615": {
        "attachment": false,
        "cost": {
          "input": 0.0374,
          "output": 0.374
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-1-6-flash-250615",
        "last_updated": "2025-06-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 1.6 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-15",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-1-6-thinking-250615": {
        "attachment": false,
        "cost": {
          "input": 0.204,
          "output": 2.04
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-1-6-thinking-250615",
        "last_updated": "2025-06-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 1.6 Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-15",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-1-8-251215": {
        "attachment": false,
        "cost": {
          "input": 0.612,
          "output": 6.12
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-1-8-251215",
        "last_updated": "2025-12-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 1.8",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-15",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-2-0-code-preview-260215": {
        "attachment": false,
        "cost": {
          "input": 0.782,
          "output": 3.893
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-2-0-code-preview-260215",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Code Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-14",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-2-0-lite-260215": {
        "attachment": false,
        "cost": {
          "input": 0.1462,
          "output": 0.8738
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-2-0-lite-260215",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-14",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-2-0-mini-260215": {
        "attachment": false,
        "cost": {
          "input": 0.0493,
          "output": 0.4845
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-2-0-mini-260215",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-14",
        "structured_output": false,
        "tool_call": false
      },
      "doubao-seed-2-0-pro-260215": {
        "attachment": false,
        "cost": {
          "input": 0.782,
          "output": 3.876
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "doubao-seed-2-0-pro-260215",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-14",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-4.5-8k-preview": {
        "attachment": false,
        "cost": {
          "input": 0.66,
          "output": 2.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-4.5-8k-preview",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 8000,
          "input": 8000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie 4.5 8k Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-25",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-4.5-turbo-128k": {
        "attachment": true,
        "cost": {
          "input": 0.132,
          "output": 0.55
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-4.5-turbo-128k",
        "last_updated": "2025-05-08",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie 4.5 Turbo 128k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-08",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-4.5-turbo-vl-32k": {
        "attachment": true,
        "cost": {
          "input": 0.495,
          "output": 1.43
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-4.5-turbo-vl-32k",
        "last_updated": "2025-05-08",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie 4.5 Turbo VL 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-08",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-5.0-thinking-preview": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-5.0-thinking-preview",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie 5.0 Thinking Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-18",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "input": 0.75,
          "output": 3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-5.1",
        "last_updated": "2026-05-10",
        "limit": {
          "context": 119000,
          "input": 119000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 5.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-10",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-5.1:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "input": 0.75,
          "output": 3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-5.1:thinking",
        "last_updated": "2026-05-10",
        "limit": {
          "context": 119000,
          "input": 119000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 5.1 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-10",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-x1-32k": {
        "attachment": true,
        "cost": {
          "input": 0.33,
          "output": 1.32
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-x1-32k",
        "last_updated": "2025-05-08",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie X1 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-08",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-x1-32k-preview": {
        "attachment": false,
        "cost": {
          "input": 0.33,
          "output": 1.32
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-x1-32k-preview",
        "last_updated": "2025-04-03",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie X1 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-03",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-x1-turbo-32k": {
        "attachment": true,
        "cost": {
          "input": 0.165,
          "output": 0.66
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-x1-turbo-32k",
        "last_updated": "2025-05-08",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ernie X1 Turbo 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-08",
        "structured_output": false,
        "tool_call": false
      },
      "ernie-x1.1-preview": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "ernie-x1.1-preview",
        "last_updated": "2025-09-10",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE X1.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-10",
        "structured_output": false,
        "tool_call": false
      },
      "essentialai/rnj-1-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "rnj",
        "id": "essentialai/rnj-1-instruct",
        "last_updated": "2025-12-13",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "RNJ-1 Instruct 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-13",
        "structured_output": false,
        "tool_call": false
      },
      "exa-answer": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "exa-answer",
        "last_updated": "2025-06-04",
        "limit": {
          "context": 4096,
          "input": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Exa (Answer)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-04",
        "structured_output": false,
        "tool_call": false
      },
      "exa-research": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "exa-research",
        "last_updated": "2025-06-04",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Exa (Research)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-04",
        "structured_output": false,
        "tool_call": false
      },
      "exa-research-pro": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "exa-research-pro",
        "last_updated": "2025-06-04",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Exa (Research Pro)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-04",
        "structured_output": false,
        "tool_call": false
      },
      "failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 70B abliterated",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "fastgpt": {
        "attachment": false,
        "cost": {
          "input": 7.5,
          "output": 7.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "fastgpt",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Web Answer",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-08-01",
        "structured_output": false,
        "tool_call": false
      },
      "featherless-ai/Qwerky-72B": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 0.5
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "qwerky",
        "id": "featherless-ai/Qwerky-72B",
        "last_updated": "2025-03-20",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwerky 72B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-20",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.0-flash-001": {
        "attachment": true,
        "cost": {
          "input": 0.1003,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.0-flash-001",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-11",
        "structured_output": true,
        "tool_call": true
      },
      "gemini-2.0-flash-exp-image-generation": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-2.0-flash-exp-image-generation",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 32767,
          "input": 32767,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Text + Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.0-flash-thinking-exp-01-21": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 1.003
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.0-flash-thinking-exp-01-21",
        "last_updated": "2025-01-21",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash Thinking 0121",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-21",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.0-flash-thinking-exp-1219": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.0-flash-thinking-exp-1219",
        "last_updated": "2024-12-19",
        "limit": {
          "context": 32767,
          "input": 32767,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash Thinking 1219",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-19",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.0-pro-exp-02-05": {
        "attachment": true,
        "cost": {
          "input": 1.989,
          "output": 7.956
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.0-pro-exp-02-05",
        "last_updated": "2025-02-05",
        "limit": {
          "context": 2097152,
          "input": 2097152,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Pro 0205",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-05",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.0-pro-reasoner": {
        "attachment": false,
        "cost": {
          "input": 1.292,
          "output": 4.998
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.0-pro-reasoner",
        "last_updated": "2025-02-05",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Pro Reasoner",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-05",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-05",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-lite",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-lite-preview-06-17": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-lite-preview-06-17",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-lite-preview-09-2025": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-lite-preview-09-2025",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite Preview (09/2025)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite-preview-09-2025-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-lite-preview-09-2025-thinking",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite Preview (09/2025) – Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-25",
        "structured_output": true,
        "tool_call": true
      },
      "gemini-2.5-flash-nothinking": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-nothinking",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash (No Thinking)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-05",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-preview-04-17": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-preview-04-17",
        "last_updated": "2025-04-17",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-17",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-preview-04-17:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 3.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-preview-04-17:thinking",
        "last_updated": "2025-04-17",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Preview Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-17",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-preview-05-20": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-preview-05-20",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 1048000,
          "input": 1048000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash 0520",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-preview-05-20:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 3.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-preview-05-20:thinking",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 1048000,
          "input": 1048000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash 0520 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-20",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-flash-preview-09-2025": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-preview-09-2025",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Preview (09/2025)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "tool_call": true
      },
      "gemini-2.5-flash-preview-09-2025-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-flash-preview-09-2025-thinking",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Preview (09/2025) – Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-25",
        "structured_output": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-pro",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-05",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-pro-exp-03-25": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-pro-exp-03-25",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro Experimental 0325",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-03-25",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-pro-preview-03-25": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-pro-preview-03-25",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro Preview 0325",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-03-25",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-pro-preview-05-06": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-pro-preview-05-06",
        "last_updated": "2025-05-06",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro Preview 0506",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-06",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-2.5-pro-preview-06-05": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-2.5-pro-preview-06-05",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro Preview 0605",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-05",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-3-pro-image-preview": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 12
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-3-pro-image-preview",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-18",
        "structured_output": false,
        "tool_call": false
      },
      "gemini-exp-1206": {
        "attachment": true,
        "cost": {
          "input": 1.258,
          "output": 4.998
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemini-exp-1206",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 2097152,
          "input": 2097152,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Pro 1206",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "gemma-4-31B-Fabled": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemma-4-31B-Fabled",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Fabled",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "gemma-4-31B-Garnet": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemma-4-31B-Garnet",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Garnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "gemma-4-31B-K1-v5": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemma-4-31B-K1-v5",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B K1 v5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "gemma-4-31B-Larkspur-v0.5": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemma-4-31B-Larkspur-v0.5",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Larkspur v0.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "gemma-4-31B-MeroMero": {
        "attachment": true,
        "cost": {
          "input": 0.306,
          "output": 0.306
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "gemma-4-31B-MeroMero",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B MeroMero",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-02",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4": {
        "attachment": false,
        "cost": {
          "input": 14.994,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4",
        "last_updated": "2024-01-16",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-16",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-air": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-air",
        "last_updated": "2024-06-05",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4 Air",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-05",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-air-0111": {
        "attachment": false,
        "cost": {
          "input": 0.1394,
          "output": 0.1394
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-air-0111",
        "last_updated": "2025-01-11",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4 Air 0111",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-11",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-airx": {
        "attachment": false,
        "cost": {
          "input": 2.006,
          "output": 2.006
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-airx",
        "last_updated": "2024-06-05",
        "limit": {
          "context": 8000,
          "input": 8000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4 AirX",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-05",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-flash": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1003
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-flash",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-long": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-long",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4 Long",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-plus": {
        "attachment": false,
        "cost": {
          "input": 7.497,
          "output": 7.497
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-plus",
        "last_updated": "2024-08-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4 Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4-plus-0111": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 9.996
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4-plus-0111",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4 Plus 0111",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4.1v-thinking-flash": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4.1v-thinking-flash",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.1V Thinking Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "tool_call": false
      },
      "glm-4.1v-thinking-flashx": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-4.1v-thinking-flashx",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.1V Thinking FlashX",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "tool_call": false
      },
      "glm-z1-air": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.07
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-z1-air",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM Z1 Air",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": true,
        "tool_call": true
      },
      "glm-z1-airx": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-z1-airx",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM Z1 AirX",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": true,
        "tool_call": true
      },
      "glm-zero-preview": {
        "attachment": false,
        "cost": {
          "input": 1.802,
          "output": 1.802
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "glm-zero-preview",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 8000,
          "input": 8000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM Zero Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": false,
        "tool_call": false
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash (Preview)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview-thinking",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-3.1-flash-lite",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro (Preview)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview-customtools",
        "last_updated": "2026-02-27",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro (Preview Custom Tools)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-27",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-high": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview-high",
        "last_updated": "2026-02-21",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro (Preview High)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-21",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-low": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview-low",
        "last_updated": "2026-02-21",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro (Preview Low)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-21",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3.5-flash",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash-thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3.5-flash-thinking",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-19",
        "structured_output": false,
        "tool_call": false
      },
      "google/gemini-flash-1.5": {
        "attachment": false,
        "cost": {
          "input": 0.0748,
          "output": 0.306
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-flash-1.5",
        "last_updated": "2024-05-14",
        "limit": {
          "context": 2000000,
          "input": 2000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 1.5 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-14",
        "structured_output": false,
        "tool_call": false
      },
      "google/gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-flash-latest",
        "last_updated": "2026-03-29",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-29",
        "structured_output": false,
        "tool_call": false
      },
      "google/gemini-flash-lite-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-flash-lite-latest",
        "last_updated": "2026-03-29",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash Lite Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-29",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemini-pro-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-pro-latest",
        "last_updated": "2026-03-29",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Pro Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-29",
        "structured_output": true,
        "tool_call": true
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "tool_call": false
      },
      "google/gemma-4-26b-a4b-it:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-26b-a4b-it:thinking",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "tool_call": false
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.35
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "tool_call": false
      },
      "google/gemma-4-31b-it:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.35
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-31b-it:thinking",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "tool_call": false
      },
      "hermes-high": {
        "attachment": true,
        "cost": {
          "input": 4.998,
          "output": 25.007
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "hermes-high",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes High",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": true,
        "tool_call": true
      },
      "hermes-low": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "hermes-low",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes Low",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": true,
        "tool_call": true
      },
      "hermes-medium": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "hermes-medium",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 204800,
          "input": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes Medium",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": true,
        "tool_call": true
      },
      "holo3-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "holo3-35b-a3b",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Holo3-35B-A3B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2024-01-01",
        "structured_output": true,
        "tool_call": true
      },
      "holo3-35b-a3b:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "holo3-35b-a3b:thinking",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Holo3-35B-A3B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-01",
        "structured_output": true,
        "tool_call": true
      },
      "huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek",
        "id": "huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Llama 70B Abliterated",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": false,
        "tool_call": false
      },
      "huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated": {
        "attachment": false,
        "cost": {
          "input": 1.4,
          "output": 1.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Qwen Abliterated",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": false,
        "tool_call": false
      },
      "huihui-ai/Llama-3.3-70B-Instruct-abliterated": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "huihui-ai/Llama-3.3-70B-Instruct-abliterated",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct abliterated",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-08",
        "structured_output": false,
        "tool_call": false
      },
      "huihui-ai/Qwen2.5-32B-Instruct-abliterated": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "huihui-ai/Qwen2.5-32B-Instruct-abliterated",
        "last_updated": "2025-01-06",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 32B Abliterated",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-06",
        "structured_output": false,
        "tool_call": false
      },
      "hunyuan-turbos-20250226": {
        "attachment": false,
        "cost": {
          "input": 0.187,
          "output": 0.374
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "hunyuan-turbos-20250226",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 24000,
          "input": 24000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hunyuan Turbo S",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": false,
        "tool_call": false
      },
      "ibm-granite/granite-4.1-8b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.05,
          "output": 0.1
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "ibm-granite/granite-4.1-8b",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Granite 4.1 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "inclusionai/ling-2.6-1t",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling 2.6 1T",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-23",
        "structured_output": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-flash": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.24
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "inclusionai/ling-2.6-flash",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling 2.6 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "structured_output": true,
        "tool_call": true
      },
      "inclusionai/ring-2.6-1t": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "inclusionai/ring-2.6-1t",
        "last_updated": "2026-05-08",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ring 2.6 1T",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-08",
        "structured_output": true,
        "tool_call": true
      },
      "inflatebot/MN-12B-Mag-Mell-R1": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "inflatebot/MN-12B-Mag-Mell-R1",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mag Mell R1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "tool_call": false
      },
      "inflection/inflection-3-pi": {
        "attachment": false,
        "cost": {
          "input": 2.499,
          "output": 9.996
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "inflection/inflection-3-pi",
        "last_updated": "2024-10-11",
        "limit": {
          "context": 8000,
          "input": 8000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inflection 3 Pi",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "structured_output": false,
        "tool_call": false
      },
      "inflection/inflection-3-productivity": {
        "attachment": false,
        "cost": {
          "input": 2.499,
          "output": 9.996
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "inflection/inflection-3-productivity",
        "last_updated": "2024-10-11",
        "limit": {
          "context": 8000,
          "input": 8000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inflection 3 Productivity",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "structured_output": false,
        "tool_call": false
      },
      "jamba-large": {
        "attachment": false,
        "cost": {
          "input": 1.989,
          "output": 7.99
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "jamba-large",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "tool_call": false
      },
      "jamba-large-1.6": {
        "attachment": false,
        "cost": {
          "input": 1.989,
          "output": 7.99
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "jamba-large-1.6",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Large 1.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": false,
        "tool_call": false
      },
      "jamba-large-1.7": {
        "attachment": false,
        "cost": {
          "input": 1.989,
          "output": 7.99
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "jamba-large-1.7",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Large 1.7",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "tool_call": false
      },
      "jamba-mini": {
        "attachment": false,
        "cost": {
          "input": 0.1989,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "jamba-mini",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "tool_call": false
      },
      "jamba-mini-1.6": {
        "attachment": false,
        "cost": {
          "input": 0.1989,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "jamba-mini-1.6",
        "last_updated": "2025-03-01",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Mini 1.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-01",
        "structured_output": false,
        "tool_call": false
      },
      "jamba-mini-1.7": {
        "attachment": false,
        "cost": {
          "input": 0.1989,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "jamba-mini-1.7",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Mini 1.7",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": false,
        "tool_call": false
      },
      "kimi-k2-instruct-fast": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "kimi-k2-instruct-fast",
        "last_updated": "2025-07-15",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0711 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-15",
        "structured_output": false,
        "tool_call": false
      },
      "kimi-thinking-preview": {
        "attachment": true,
        "cost": {
          "input": 31.46,
          "output": 31.46
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "kimi-thinking-preview",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi Thinking Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "structured_output": false,
        "tool_call": false
      },
      "kwaipilot/kat-coder-pro-v2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "kwaipilot/kat-coder-pro-v2",
        "last_updated": "2026-03-28",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KAT Coder Pro V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-28",
        "structured_output": false,
        "tool_call": false
      },
      "learnlm-1.5-pro-experimental": {
        "attachment": false,
        "cost": {
          "input": 3.502,
          "output": 10.506
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "learnlm-1.5-pro-experimental",
        "last_updated": "2024-05-14",
        "limit": {
          "context": 32767,
          "input": 32767,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini LearnLM Experimental",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-14",
        "structured_output": false,
        "tool_call": false
      },
      "liquid/lfm-2-24b-a2b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.12
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "liquid/lfm-2-24b-a2b",
        "last_updated": "2025-12-20",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LFM2 24B A2B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-20",
        "structured_output": false,
        "tool_call": false
      },
      "meganova-ai/manta-flash-1.0": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.16
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova",
        "id": "meganova-ai/manta-flash-1.0",
        "last_updated": "2025-12-20",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Manta Flash 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-20",
        "structured_output": false,
        "tool_call": false
      },
      "meganova-ai/manta-mini-1.0": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.16
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova",
        "id": "meganova-ai/manta-mini-1.0",
        "last_updated": "2025-12-20",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Manta Mini 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-20",
        "structured_output": false,
        "tool_call": false
      },
      "meganova-ai/manta-pro-1.0": {
        "attachment": false,
        "cost": {
          "input": 0.060000000000000005,
          "output": 0.5
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova",
        "id": "meganova-ai/manta-pro-1.0",
        "last_updated": "2025-12-20",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Manta Pro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-20",
        "structured_output": false,
        "tool_call": false
      },
      "mercury-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "mercury-2",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-01",
        "structured_output": true,
        "tool_call": true
      },
      "meta-llama/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.0544,
          "output": 0.0544
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.1-8b-instruct",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8b Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": false,
        "tool_call": false
      },
      "meta-llama/llama-3.2-3b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.0306,
          "output": 0.0493
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta-llama/llama-3.2-3b-instruct",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3b Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "meta-llama/llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.23
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.3-70b-instruct",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70b Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": true,
        "tool_call": true
      },
      "meta-llama/llama-4-maverick": {
        "attachment": true,
        "cost": {
          "input": 0.18000000000000002,
          "output": 0.8
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta-llama/llama-4-maverick",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "structured_output": true,
        "tool_call": true
      },
      "meta-llama/llama-4-scout": {
        "attachment": true,
        "cost": {
          "input": 0.085,
          "output": 0.46
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta-llama/llama-4-scout",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 328000,
          "input": 328000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "structured_output": true,
        "tool_call": true
      },
      "microsoft/wizardlm-2-8x22b": {
        "attachment": true,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "microsoft/wizardlm-2-8x22b",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "WizardLM-2 8x22B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "minimax/minimax-01": {
        "attachment": true,
        "cost": {
          "input": 0.1394,
          "output": 1.1219999999999999
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "minimax/minimax-01",
        "last_updated": "2025-01-15",
        "limit": {
          "context": 1000192,
          "input": 1000192,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax 01",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-15",
        "structured_output": false,
        "tool_call": false
      },
      "minimax/minimax-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "id": "minimax/minimax-latest",
        "last_updated": "2026-05-03",
        "limit": {
          "context": 512000,
          "input": 512000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-03",
        "structured_output": true,
        "tool_call": true
      },
      "minimax/minimax-m2-her": {
        "attachment": false,
        "cost": {
          "input": 0.30200000000000005,
          "output": 1.2069999999999999
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2-her",
        "last_updated": "2026-01-24",
        "limit": {
          "context": 65532,
          "input": 65532,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2-her",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-24",
        "structured_output": false,
        "tool_call": false
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "input": 0.33,
          "output": 1.32
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.1",
        "last_updated": "2025-12-19",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-19",
        "structured_output": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "input": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "input": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "Efficient MiniMax model for quick assistance, coding, and routine automation",
        "id": "minimax/minimax-m2.7-turbo",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "input": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7 Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "tool_call": true
      },
      "minimax/minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "id": "minimax/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "input": 512000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-01",
        "structured_output": true,
        "tool_call": true
      },
      "minimax/minimax-m3:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "id": "minimax/minimax-m3:thinking",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "input": 512000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M3 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "structured_output": true,
        "tool_call": true
      },
      "mirothinker-1-7-deepresearch": {
        "attachment": false,
        "cost": {
          "input": 4,
          "output": 25
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "id": "mirothinker-1-7-deepresearch",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiroThinker 1.7 Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": false,
        "tool_call": false
      },
      "mirothinker-1-7-deepresearch-mini": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "id": "mirothinker-1-7-deepresearch-mini",
        "last_updated": "2026-05-11",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiroThinker 1.7 Deep Research Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-11",
        "structured_output": false,
        "tool_call": false
      },
      "mistral-code-agent-latest": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "mistral-code-agent-latest",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Code Agent Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-02",
        "structured_output": true,
        "tool_call": true
      },
      "mistral-code-latest": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "mistral-code-latest",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Code Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-02",
        "structured_output": true,
        "tool_call": true
      },
      "mistral-small-31-24b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "mistral-small-31-24b-instruct",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 31 24b Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "mistral/mistral-medium-3.5": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistral/mistral-medium-3.5",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "mistral/mistral-medium-3.5:thinking": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistral/mistral-medium-3.5:thinking",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-30",
        "structured_output": true,
        "tool_call": true
      },
      "mistralai/Devstral-Small-2505": {
        "attachment": false,
        "cost": {
          "input": 0.060000000000000005,
          "output": 0.060000000000000005
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistralai/Devstral-Small-2505",
        "last_updated": "2025-08-02",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Devstral Small 2505",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-02",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/Mistral-Nemo-Instruct-2407": {
        "attachment": false,
        "cost": {
          "input": 0.1003,
          "output": 0.1207
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistralai/Mistral-Nemo-Instruct-2407",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/codestral-2508": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.8999999999999999
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "codestral",
        "id": "mistralai/codestral-2508",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral 2508",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-01",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/devstral-2-123b-instruct-2512": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.4
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistralai/devstral-2-123b-instruct-2512",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2 123B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-09",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/ministral-14b-2512": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-14b-2512",
        "last_updated": "2025-12-04",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 14B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-04",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/ministral-14b-instruct-2512": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-14b-instruct-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 14B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/ministral-3b-2512": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-3b-2512",
        "last_updated": "2025-12-04",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-04",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/ministral-8b-2512": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-8b-2512",
        "last_updated": "2025-12-04",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-04",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-large": {
        "attachment": false,
        "cost": {
          "input": 2.006,
          "output": 6.001
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistralai/mistral-large",
        "last_updated": "2024-02-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 2411",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-02-26",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-large-3-675b-instruct-2512": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistralai/mistral-large-3-675b-instruct-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3 675B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-medium-3": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/mistral-medium-3",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-medium-3.1": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/mistral-medium-3.1",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-saba": {
        "attachment": false,
        "cost": {
          "input": 0.1989,
          "output": 0.595
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "mistralai/mistral-saba",
        "last_updated": "2025-02-17",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Saba",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-17",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-small-4-119b-2603": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.4
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-4-119b-2603",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 4 119B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "tool_call": true
      },
      "mistralai/mistral-small-4-119b-2603:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.4
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-4-119b-2603:thinking",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 4 119B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-17",
        "structured_output": true,
        "tool_call": true
      },
      "mistralai/mixtral-8x22b-instruct-v0.1": {
        "attachment": false,
        "cost": {
          "input": 0.8999999999999999,
          "output": 0.8999999999999999
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mixtral",
        "id": "mistralai/mixtral-8x22b-instruct-v0.1",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mixtral 8x22B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mixtral-8x7b-instruct-v0.1": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.27
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mixtral",
        "id": "mistralai/mixtral-8x7b-instruct-v0.1",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mixtral 8x7B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "structured_output": false,
        "tool_call": false
      },
      "mlabonne/NeuralDaredevil-8B-abliterated": {
        "attachment": false,
        "cost": {
          "input": 0.44,
          "output": 0.44
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "mlabonne/NeuralDaredevil-8B-abliterated",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Neural Daredevil 8B abliterated",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "structured_output": false,
        "tool_call": false
      },
      "moonshotai/Kimi-K2-Instruct-0905": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2-Instruct-0905",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-instruct",
        "last_updated": "2025-07-01",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-01",
        "structured_output": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-instruct-0711": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 2
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-instruct-0711",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0711",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-11",
        "structured_output": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-06",
        "structured_output": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking-original": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking-original",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "structured_output": false,
        "tool_call": false
      },
      "moonshotai/kimi-k2-thinking-turbo-original": {
        "attachment": false,
        "cost": {
          "input": 1.15,
          "output": 8
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking-turbo-original",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking Turbo Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "structured_output": false,
        "tool_call": false
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.9
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "last_updated": "2026-01-26",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-26",
        "structured_output": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.9
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2.5:thinking",
        "last_updated": "2026-01-26",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-26",
        "structured_output": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "input": 0.53,
          "output": 2.73
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-16",
        "structured_output": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.53,
          "output": 2.73
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2.6:thinking",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": false,
        "tool_call": true
      },
      "moonshotai/kimi-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 0.5,
          "output": 2.6
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "moonshotai/kimi-latest",
        "last_updated": "2026-05-03",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-03",
        "structured_output": false,
        "tool_call": true
      },
      "nanogpt/coding-router": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 2.2
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "nanogpt/coding-router",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "tool_call": true
      },
      "nanogpt/coding-router:high": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 2.2
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "nanogpt/coding-router:high",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Router High",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "tool_call": true
      },
      "nanogpt/coding-router:low": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "nanogpt/coding-router:low",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Router Low",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "tool_call": true
      },
      "nanogpt/coding-router:max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "nanogpt/coding-router:max",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Router Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "tool_call": true
      },
      "nanogpt/coding-router:medium": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "nanogpt/coding-router:medium",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coding Router Medium",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "tool_call": true
      },
      "nex-agi/deepseek-v3.1-nex-n1": {
        "attachment": false,
        "cost": {
          "input": 0.27999999999999997,
          "output": 0.42000000000000004
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "nex-agi/deepseek-v3.1-nex-n1",
        "last_updated": "2025-12-10",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Nex N1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-10",
        "structured_output": false,
        "tool_call": false
      },
      "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16": {
        "attachment": false,
        "cost": {
          "input": 0.49299999999999994,
          "output": 0.49299999999999994
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Celeste v0.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": false,
        "tool_call": false
      },
      "nvidia/Llama-3.1-Nemotron-70B-Instruct-HF": {
        "attachment": false,
        "cost": {
          "input": 0.357,
          "output": 0.408
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia/Llama-3.1-Nemotron-70B-Instruct-HF",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron 70b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nvidia/Llama-3.3-Nemotron-Super-49B-v1": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron Super 49B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nvidia/Llama-3_3-Nemotron-Super-49B-v1_5": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia/Llama-3_3-Nemotron-Super-49B-v1_5",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron Super 49B v1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nvidia/nemotron-3-nano-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-30b-a3b",
        "last_updated": "2025-12-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron 3 Nano 30B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0.105,
          "output": 0.42
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning",
        "last_updated": "2026-04-28",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron 3 Nano Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron 3 Super 120B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b:thinking",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron 3 Super 120B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nvidia-nemotron-nano-9b-v2": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia/nvidia-nemotron-nano-9b-v2",
        "last_updated": "2025-08-18",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron Nano 9B v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-18",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 16385,
          "input": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2022-11-30",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4-turbo-preview": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 30.004999999999995
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo-preview",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "last_updated": "2025-09-10",
        "limit": {
          "context": 1047576,
          "input": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-10",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "input": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 4.1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "input": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 4.1 Nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "input": 2.499,
          "output": 9.996
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4o-2024-08-06": {
        "attachment": true,
        "cost": {
          "input": 2.499,
          "output": 9.996
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-08-06",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-08-06)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-06",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-11-20",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "input": 0.1496,
          "output": 0.595
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4o-mini-search-preview": {
        "attachment": false,
        "cost": {
          "input": 0.088,
          "output": 0.35
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini-search-preview",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-4o-search-preview": {
        "attachment": true,
        "cost": {
          "input": 1.47,
          "output": 5.88
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-search-preview",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5-codex": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5-codex",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Codex",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-15",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai/gpt-5-pro",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.1-2025-11-13": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1-2025-11-13",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 (2025-11-13)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-13",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 20
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-max",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex-mini",
        "id": "openai/gpt-5.1-codex-mini",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.1 Codex Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "last_updated": "2026-01-01",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-01",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-codex",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-14",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai/gpt-5.2-pro",
        "last_updated": "2026-01-01",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-01",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.3-codex",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.4",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 922000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-mini",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-nano",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 3,
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.4-pro",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 922000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.5",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-chat-latest",
        "last_updated": "2026-05-03",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT Chat Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-03",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-latest",
        "last_updated": "2026-03-29",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-29",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.15
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "tool_call": false
      },
      "openai/gpt-oss-safeguard-20b": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-safeguard-20b",
        "last_updated": "2025-10-29",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS Safeguard 20B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-29",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o1": {
        "attachment": false,
        "cost": {
          "input": 14.993999999999998,
          "output": 59.993
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "last_updated": "2024-12-17",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o1-preview": {
        "attachment": false,
        "cost": {
          "input": 14.993999999999998,
          "output": 59.993
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1-preview",
        "last_updated": "2024-09-12",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o1-preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-09-12",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o1-pro": {
        "attachment": true,
        "cost": {
          "input": 150,
          "output": 600
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "openai/o1-pro",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o1 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-25",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o3": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o3",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-16",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o3-deep-research": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o",
        "id": "openai/o3-deep-research",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3 Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o3-mini": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-01-31",
        "structured_output": true,
        "tool_call": true
      },
      "openai/o3-mini-high": {
        "attachment": false,
        "cost": {
          "input": 0.64,
          "output": 2.588
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini-high",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3-mini (High)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "structured_output": true,
        "tool_call": true
      },
      "openai/o3-mini-low": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini-low",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3-mini (Low)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "structured_output": true,
        "tool_call": true
      },
      "openai/o3-pro-2025-06-10": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "openai/o3-pro-2025-06-10",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o3-pro (2025-06-10)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "structured_output": true,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "tool_call": true
      },
      "openai/o4-mini-deep-research": {
        "attachment": false,
        "cost": {
          "input": 9.996,
          "output": 19.992
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o-mini",
        "id": "openai/o4-mini-deep-research",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o4-mini Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": false,
        "tool_call": false
      },
      "openai/o4-mini-high": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o4-mini-high",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI o4-mini high",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-16",
        "structured_output": true,
        "tool_call": true
      },
      "owl": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "owl",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 1048756,
          "input": 1048756,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OWL",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-01",
        "structured_output": true,
        "tool_call": true
      },
      "pamanseau/OpenReasoning-Nemotron-32B": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "pamanseau/OpenReasoning-Nemotron-32B",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenReasoning Nemotron 32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-21",
        "structured_output": false,
        "tool_call": false
      },
      "perceptron/perceptron-mk1": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "perceptron/perceptron-mk1",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perceptron Mk1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "tool_call": false
      },
      "phi-4-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.68
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "phi-4-mini-instruct",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi 4 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "phi-4-multimodal-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.11
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "phi-4-multimodal-instruct",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi 4 Multimodal",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "poolside/laguna-m.1": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "poolside/laguna-m.1",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna M.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-29",
        "structured_output": false,
        "tool_call": false
      },
      "poolside/laguna-xs.2": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "poolside/laguna-xs.2",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna XS.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-29",
        "structured_output": false,
        "tool_call": false
      },
      "qvq-max": {
        "attachment": true,
        "cost": {
          "input": 1.4,
          "output": 5.3
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qvq-max",
        "last_updated": "2025-03-28",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: QvQ Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-28",
        "structured_output": false,
        "tool_call": false
      },
      "qwen-3.6-plus": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 2.7
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "qwen3.6",
        "id": "qwen-3.6-plus",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 991800,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-02",
        "structured_output": false,
        "tool_call": false
      },
      "qwen-long": {
        "attachment": true,
        "cost": {
          "input": 0.1003,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen-long",
        "last_updated": "2025-01-25",
        "limit": {
          "context": 10000000,
          "input": 10000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Long 10M",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-25",
        "structured_output": false,
        "tool_call": false
      },
      "qwen-max": {
        "attachment": false,
        "cost": {
          "input": 1.5997,
          "output": 6.392
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen-max",
        "last_updated": "2024-04-03",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-03",
        "structured_output": false,
        "tool_call": false
      },
      "qwen-plus": {
        "attachment": false,
        "cost": {
          "input": 0.3995,
          "output": 1.2002
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen-plus",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 995904,
          "input": 995904,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2024-01-25",
        "structured_output": false,
        "tool_call": false
      },
      "qwen-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.04998,
          "output": 0.2006
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen-turbo",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/Qwen2.5-Coder-32B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/Qwen2.5-Coder-32B-Instruct",
        "last_updated": "2025-07-03",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 Coder 32b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-03",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/Qwen3-235B-A22B-Instruct-2507",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235b A22B 2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-25",
        "structured_output": true,
        "tool_call": true
      },
      "qwen/Qwen3-235B-A22B-Instruct-2507-TEE": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/Qwen3-235B-A22B-Instruct-2507-TEE",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235b A22B 2507 (TEE)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-25",
        "structured_output": true,
        "tool_call": true
      },
      "qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.5
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/Qwen3-235B-A22B-Thinking-2507",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235b A22B 2507 Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-11",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/Qwen3-8B": {
        "attachment": false,
        "cost": {
          "input": 0.47,
          "output": 0.47
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/Qwen3-8B",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 41000,
          "input": 41000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/Qwen3-Next-80B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.65
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/Qwen3-Next-80B-A3B-Instruct",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B (Instruct)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-11",
        "structured_output": true,
        "tool_call": true
      },
      "qwen/Qwen3-VL-235B-A22B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/Qwen3-VL-235B-A22B-Instruct",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/Qwen3.6-35B-A3B": {
        "attachment": true,
        "cost": {
          "input": 0.112,
          "output": 0.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/Qwen3.6-35B-A3B",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-17",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/Qwen3.6-35B-A3B:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.112,
          "output": 0.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/Qwen3.6-35B-A3B:thinking",
        "last_updated": "2026-04-19",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-19",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen-2.5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.357,
          "output": 0.408
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen-2.5-72b-instruct",
        "last_updated": "2025-07-03",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 72B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-03",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-14b": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.24
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-14b",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 41000,
          "input": 41000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 14b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-235b-a22b": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-235b-a22b",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 41000,
          "input": 41000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235b A22B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-30b-a3b",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 41000,
          "input": 41000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-32b": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-32b",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 41000,
          "input": 41000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 32b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-coder": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 262000,
          "input": 262000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Coder 480B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-17",
        "structured_output": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-flash",
        "last_updated": "2025-09-17",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-17",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-next",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-08",
        "structured_output": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-plus",
        "last_updated": "2025-09-17",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-17",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 1.08018,
          "output": 5.4009
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "id": "qwen/qwen3-max",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.65
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-next-80b-a3b-thinking",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B (Thinking)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-397b-a17b",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 258048,
          "input": 258048,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-16",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3.5-397b-a17b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-397b-a17b-thinking",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 258048,
          "input": 258048,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3.5-9b": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.15
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-9b",
        "last_updated": "2026-03-10",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-10",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3.5-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-plus",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 983616,
          "input": 983616,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-16",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwen3.5-plus-thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-plus-thinking",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 983616,
          "input": 983616,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "structured_output": false,
        "tool_call": false
      },
      "qwen/qwq-32b-preview": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwq-32b-preview",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen QwQ 32B Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "structured_output": false,
        "tool_call": false
      },
      "qwen25-vl-72b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.69989,
          "output": 0.69989
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen25-vl-72b-instruct",
        "last_updated": "2025-05-10",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen25 VL 72b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-10",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3-30b-a3b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3-30b-a3b-instruct-2507",
        "last_updated": "2025-02-20",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct 2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-20",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3-coder-30b-a3b-instruct",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30B A3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": true,
        "tool_call": true
      },
      "qwen3-max-2026-01-23": {
        "attachment": false,
        "cost": {
          "input": 1.2002,
          "output": 6.001
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3-max-2026-01-23",
        "last_updated": "2026-01-26",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max 2026-01-23",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-26",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3-vl-235b-a22b-instruct-original": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3-vl-235b-a22b-instruct-original",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct Original",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3-vl-235b-a22b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3-vl-235b-a22b-thinking",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-26",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0.36,
          "output": 2.88
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-122b-a10b",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B A10B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-122b-a10b:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.36,
          "output": 2.88
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-122b-a10b:thinking",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B A10B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.27,
          "output": 2.16
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-27b",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-27b:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.27,
          "output": 2.16
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-27b:thinking",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.225,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-35b-a3b",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-35b-a3b:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.225,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-35b-a3b:thinking",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 260096,
          "input": 260096,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B A3B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-flash",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 991808,
          "input": 991808,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-flash:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.09,
          "output": 0.36
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.5-flash:thinking",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 991808,
          "input": 991808,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Flash Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-omni-flash": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks",
        "id": "qwen3.5-omni-flash",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 49152,
          "input": 49152,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Omni Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.5-omni-plus": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Omni-modal model for text, vision, audio, and multimodal agent tasks",
        "id": "qwen3.5-omni-plus",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 983616,
          "input": 983616,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Omni Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "input": 1.3,
          "output": 7.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "qwen3.6",
        "id": "qwen3.6-max-preview",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 245800,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-20",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-21",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.7-max:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.7-max:thinking",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 262144,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.7-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.7-plus",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 991808,
          "input": 991808,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwen3.7-plus:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwen3.7-plus:thinking",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 983616,
          "input": 983616,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 262144,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-01",
        "structured_output": false,
        "tool_call": false
      },
      "qwq-32b": {
        "attachment": false,
        "cost": {
          "input": 0.25599999,
          "output": 0.30499999
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "qwq-32b",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen: QwQ 32B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "sarvam-105b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.045,
          "output": 0.177
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "sarvam-105b",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sarvam 105B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": false,
        "tool_call": true
      },
      "sarvam-30b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.017,
          "input": 0.028,
          "output": 0.111
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "sarvam-30b",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sarvam 30B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": false,
        "tool_call": true
      },
      "shisa-ai/shisa-v2-llama3.3-70b": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 0.5
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "shisa-ai/shisa-v2-llama3.3-70b",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Shisa V2 Llama 3.3 70B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "shisa-ai/shisa-v2.1-llama3.3-70b": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 0.5
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "shisa-ai/shisa-v2.1-llama3.3-70b",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Shisa V2.1 Llama 3.3 70B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "tool_call": false
      },
      "sonar": {
        "attachment": false,
        "cost": {
          "input": 1.003,
          "output": 1.003
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "sonar",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 127000,
          "input": 127000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Simple",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "structured_output": false,
        "tool_call": false
      },
      "sonar-deep-research": {
        "attachment": false,
        "cost": {
          "input": 3.4,
          "output": 13.6
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "id": "sonar-deep-research",
        "last_updated": "2025-02-25",
        "limit": {
          "context": 60000,
          "input": 60000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Deep Research",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-25",
        "structured_output": false,
        "tool_call": false
      },
      "sonar-pro": {
        "attachment": false,
        "cost": {
          "input": 2.992,
          "output": 14.994
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "sonar-pro",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "structured_output": false,
        "tool_call": false
      },
      "sonar-reasoning-pro": {
        "attachment": false,
        "cost": {
          "input": 2.006,
          "output": 7.9985
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "id": "sonar-reasoning-pro",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 127000,
          "input": 127000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-19",
        "structured_output": false,
        "tool_call": false
      },
      "soob3123/GrayLine-Qwen3-8B": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "soob3123/GrayLine-Qwen3-8B",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grayline Qwen3 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "soob3123/Veiled-Calla-12B": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "soob3123/Veiled-Calla-12B",
        "last_updated": "2025-04-13",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Veiled Calla 12B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-13",
        "structured_output": false,
        "tool_call": false
      },
      "soob3123/amoral-gemma3-27B-v2": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "soob3123/amoral-gemma3-27B-v2",
        "last_updated": "2025-05-23",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Amoral Gemma3 27B v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-23",
        "structured_output": false,
        "tool_call": false
      },
      "step-2-16k-exp": {
        "attachment": false,
        "cost": {
          "input": 7.004,
          "output": 19.992
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "step-2-16k-exp",
        "last_updated": "2024-07-05",
        "limit": {
          "context": 16000,
          "input": 16000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step-2 16k Exp",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-05",
        "structured_output": false,
        "tool_call": false
      },
      "step-2-mini": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.408
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "step-2-mini",
        "last_updated": "2024-07-05",
        "limit": {
          "context": 8000,
          "input": 8000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step-2 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-05",
        "structured_output": false,
        "tool_call": false
      },
      "step-3": {
        "attachment": true,
        "cost": {
          "input": 0.2499,
          "output": 0.6494
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "step-3",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 65536,
          "input": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step-3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-31",
        "structured_output": false,
        "tool_call": false
      },
      "step-r1-v-mini": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 11
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "step-r1-v-mini",
        "last_updated": "2025-04-08",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step R1 V Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-08",
        "structured_output": false,
        "tool_call": false
      },
      "stepfun-ai/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.5
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "family": "step",
        "id": "stepfun-ai/step-3.5-flash",
        "last_updated": "2026-02-02",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-02",
        "structured_output": false,
        "tool_call": false
      },
      "stepfun-ai/step-3.5-flash-2603": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun-ai/step-3.5-flash-2603",
        "last_updated": "2026-04-14",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash 2603",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-14",
        "structured_output": false,
        "tool_call": false
      },
      "stepfun/step-3.7-flash:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.2,
          "output": 1.15
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun/step-3.7-flash:thinking",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-29",
        "structured_output": true,
        "tool_call": true
      },
      "tencent/Hunyuan-MT-7B": {
        "attachment": false,
        "cost": {
          "input": 10,
          "output": 20
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "family": "hunyuan",
        "id": "tencent/Hunyuan-MT-7B",
        "last_updated": "2025-09-18",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hunyuan MT 7B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": false,
        "tool_call": false
      },
      "tencent/hy3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.029,
          "input": 0.066,
          "output": 0.26
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "id": "tencent/hy3-preview",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tencent: Hy3 preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-23",
        "structured_output": false,
        "tool_call": false
      },
      "undi95/remm-slerp-l2-13b": {
        "attachment": true,
        "cost": {
          "input": 0.7989999999999999,
          "output": 1.2069999999999999
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "undi95/remm-slerp-l2-13b",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 6144,
          "input": 6144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ReMM SLERP 13B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "universal-summarizer": {
        "attachment": false,
        "cost": {
          "input": 30,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "universal-summarizer",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Universal Summarizer",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "unsloth/gemma-3-12b-it": {
        "attachment": true,
        "cost": {
          "input": 0.272,
          "output": 0.272
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "unsloth",
        "id": "unsloth/gemma-3-12b-it",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 12B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-10",
        "structured_output": false,
        "tool_call": false
      },
      "unsloth/gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.2992,
          "output": 0.2992
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "unsloth",
        "id": "unsloth/gemma-3-27b-it",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 96000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 27B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-10",
        "structured_output": false,
        "tool_call": false
      },
      "unsloth/gemma-3-4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "unsloth",
        "id": "unsloth/gemma-3-4b-it",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 4B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-10",
        "structured_output": false,
        "tool_call": false
      },
      "upstage/solar-pro-3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "upstage/solar-pro-3",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Solar Pro 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": false,
        "tool_call": false
      },
      "v0-1.0-md": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "v0-1.0-md",
        "last_updated": "2025-07-04",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "v0 1.0 MD",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-04",
        "structured_output": false,
        "tool_call": false
      },
      "v0-1.5-lg": {
        "attachment": false,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "v0-1.5-lg",
        "last_updated": "2025-07-04",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "v0 1.5 LG",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-04",
        "structured_output": false,
        "tool_call": false
      },
      "v0-1.5-md": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "v0-1.5-md",
        "last_updated": "2025-07-04",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "v0 1.5 MD",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-04",
        "structured_output": false,
        "tool_call": false
      },
      "venice-uncensored": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "venice-uncensored",
        "last_updated": "2025-02-24",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Venice Uncensored",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-24",
        "structured_output": false,
        "tool_call": false
      },
      "venice-uncensored:web": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "venice-uncensored:web",
        "last_updated": "2024-05-01",
        "limit": {
          "context": 80000,
          "input": 80000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Venice Uncensored Web",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-01",
        "structured_output": false,
        "tool_call": false
      },
      "x-ai/grok-4.20": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4.20",
        "last_updated": "2026-03-31",
        "limit": {
          "context": 2000000,
          "input": 2000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-31",
        "structured_output": true,
        "tool_call": true
      },
      "x-ai/grok-4.20-multi-agent": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4.20-multi-agent",
        "last_updated": "2026-03-31",
        "limit": {
          "context": 2000000,
          "input": 2000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Multi-Agent",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-31",
        "structured_output": true,
        "tool_call": true
      },
      "x-ai/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4.3",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": true,
        "tool_call": true
      },
      "x-ai/grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 2
        },
        "description": "Grok coding model for agentic engineering, edits, and codebase workflows",
        "id": "x-ai/grok-build-0.1",
        "last_updated": "2026-05-20",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-20",
        "structured_output": true,
        "tool_call": true
      },
      "x-ai/grok-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-latest",
        "last_updated": "2026-05-03",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-03",
        "structured_output": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "input": 0.102,
          "output": 0.306
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "xiaomi/mimo-v2-flash-original": {
        "attachment": false,
        "cost": {
          "input": 0.102,
          "output": 0.306
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash-original",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "xiaomi/mimo-v2-flash-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.102,
          "output": 0.306
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash-thinking",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "xiaomi/mimo-v2-flash-thinking-original": {
        "attachment": false,
        "cost": {
          "input": 0.102,
          "output": 0.306
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash-thinking-original",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash (Thinking) Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": false,
        "tool_call": false
      },
      "xiaomi/mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 2
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "id": "xiaomi/mimo-v2-omni",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-19",
        "structured_output": false,
        "tool_call": true
      },
      "xiaomi/mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3
        },
        "description": "MiMo pro model for strong multimodal reasoning and agent execution",
        "id": "xiaomi/mimo-v2-pro",
        "last_updated": "2026-03-19",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-19",
        "structured_output": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "id": "xiaomi/mimo-v2.5",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0036,
          "input": 0.435,
          "output": 0.87
        },
        "description": "MiMo pro model for strong multimodal reasoning and agent execution",
        "id": "xiaomi/mimo-v2.5-pro",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "input": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "tool_call": true
      },
      "yi-large": {
        "attachment": false,
        "cost": {
          "input": 3.196,
          "output": 3.196
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "yi-large",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 32000,
          "input": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Yi Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": false,
        "tool_call": false
      },
      "yi-lightning": {
        "attachment": false,
        "cost": {
          "input": 0.2006,
          "output": 0.2006
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "yi-lightning",
        "last_updated": "2024-10-16",
        "limit": {
          "context": 12000,
          "input": 12000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Yi Lightning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-16",
        "structured_output": false,
        "tool_call": false
      },
      "yi-medium-200k": {
        "attachment": false,
        "cost": {
          "input": 2.499,
          "output": 2.499
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "yi-medium-200k",
        "last_updated": "2024-03-01",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Yi Medium 200k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-01",
        "structured_output": false,
        "tool_call": false
      },
      "z-ai/glm-4.5v": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 1.7999999999999998
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glmv",
        "id": "z-ai/glm-4.5v",
        "last_updated": "2025-11-22",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 96000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5V",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-11-22",
        "structured_output": false,
        "tool_call": false
      },
      "z-ai/glm-4.5v:thinking": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 1.7999999999999998
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glmv",
        "id": "z-ai/glm-4.5v:thinking",
        "last_updated": "2025-11-22",
        "limit": {
          "context": 64000,
          "input": 64000,
          "output": 96000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5V Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-22",
        "structured_output": false,
        "tool_call": false
      },
      "z-ai/glm-4.6": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "z-ai/glm-4.6",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "structured_output": true,
        "tool_call": true
      },
      "z-ai/glm-4.6:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "z-ai/glm-4.6:thinking",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "structured_output": true,
        "tool_call": true
      },
      "z-ai/glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-5-turbo",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 202800,
          "input": 202800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-15",
        "structured_output": true,
        "tool_call": true
      },
      "z-ai/glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-5v-turbo",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 202800,
          "input": 202800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5V Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-01",
        "structured_output": true,
        "tool_call": true
      },
      "z-ai/glm-5v-turbo:thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-5v-turbo:thinking",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 202800,
          "input": 202800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5V Turbo Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-Air": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.8
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/GLM-4.5-Air",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-Air:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.8
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/GLM-4.5-Air:thinking",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-01",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.3
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/GLM-4.5:thinking",
        "last_updated": "2024-01-01",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-01-01",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/GLM-4.6-turbo": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/GLM-4.6-turbo",
        "last_updated": "2025-10-02",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 204800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-02",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/GLM-4.6-turbo:thinking": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/GLM-4.6-turbo:thinking",
        "last_updated": "2025-10-02",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 204800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6 Turbo (Thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-02",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.3
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-4.5",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.6-original": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 1.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-4.6-original",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6 Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.6v": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "zai-org/glm-4.6v",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.6v-flash-original": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "zai-org/glm-4.6v-flash-original",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-08",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.6v-original": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "zai-org/glm-4.6v-original",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V Original",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-08",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.7": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.8
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.7",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-29",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "zai-org/glm-4.7-flash",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-flash-original": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/glm-4.7-flash-original",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-flash-original:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/glm-4.7-flash-original:thinking",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash Original Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.7-flash:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "zai-org/glm-4.7-flash:thinking",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "structured_output": false,
        "tool_call": false
      },
      "zai-org/glm-4.7-original": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-4.7-original",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-original:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-4.7-original:thinking",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Original Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-4.7:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-4.7:thinking",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2.55
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-5-original": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-5-original",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Original",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-5-original:thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-5-original:thinking",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Original Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-5.1": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2.55
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5.1",
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-5.1:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2.55
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5.1:thinking",
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-27",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-5:thinking": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 2.55
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5:thinking",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "structured_output": true,
        "tool_call": true
      },
      "zai-org/glm-latest": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.75,
          "output": 2.6
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "zai-org/glm-latest",
        "last_updated": "2026-05-03",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-03",
        "structured_output": true,
        "tool_call": true
      }
    },
    "name": "NanoGPT",
    "npm": "@ai-sdk/openai-compatible"
  },
  "nearai": {
    "api": "https://cloud-api.near.ai/v1",
    "doc": "https://docs.near.ai/",
    "env": [
      "NEARAI_API_KEY"
    ],
    "id": "nearai",
    "models": {
      "Qwen/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.55
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B-A3B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Embedding-0.6B": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "Qwen/Qwen3-Embedding-0.6B",
        "last_updated": "2025-06-03",
        "limit": {
          "context": 40960,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 0.6B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-03",
        "temperature": false,
        "tool_call": false
      },
      "Qwen/Qwen3-Reranker-0.6B": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0.01
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "qwen",
        "id": "Qwen/Qwen3-Reranker-0.6B",
        "last_updated": "2025-06-03",
        "limit": {
          "context": 40960,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Reranker 0.6B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-03",
        "temperature": false,
        "tool_call": false
      },
      "Qwen/Qwen3-VL-30B-A3B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.55
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-30B-A3B-Instruct",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-122B-A10B": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-122B-A10B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B-FP8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.056,
          "input": 0.17,
          "output": 1.1
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-35B-A3B-FP8",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 35B A3B FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15.5
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "black-forest-labs/FLUX.2-klein-4B": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "black-forest-labs/FLUX.2-klein-4B",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 Klein 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-14",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "input_audio": 0.3,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 1.25,
          "output": 15
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "input_audio": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.026,
          "input": 0.13,
          "output": 0.4
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.18,
          "input": 1.8,
          "output": 15.5
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.55
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/whisper-large-v3": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "openai/whisper-large-v3",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 448,
          "output": 448
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper Large v3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": false,
        "tool_call": false
      },
      "zai-org/GLM-5.1-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 3.3
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5.1-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1 FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "NEAR AI Cloud",
    "npm": "@ai-sdk/openai-compatible"
  },
  "nebius": {
    "api": "https://api.tokenfactory.nebius.com/v1",
    "doc": "https://docs.tokenfactory.nebius.com/",
    "env": [
      "NEBIUS_API_KEY"
    ],
    "id": "nebius",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 196608,
          "input": 190000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.5-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "MiniMaxAI/MiniMax-M2.5-fast",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 8000,
          "input": 7000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "NousResearch/Hermes-4-405B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 3,
          "reasoning": 3
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "NousResearch/Hermes-4-405B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-11",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes-4-405B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "NousResearch/Hermes-4-70B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.013,
          "cache_write": 0.16,
          "input": 0.13,
          "output": 0.4,
          "reasoning": 0.4
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "NousResearch/Hermes-4-70B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-11",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes-4-70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "PrimeIntellect/INTELLECT-3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.25,
          "input": 0.2,
          "output": 1.1
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "PrimeIntellect/INTELLECT-3",
        "knowledge": "2025-10",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "INTELLECT-3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-25",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-VL-72B-Instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.31,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "Qwen/Qwen2.5-VL-72B-Instruct",
        "knowledge": "2024-12",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL-72B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "knowledge": "2025-07",
        "last_updated": "2025-10-04",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "input": 0.5,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507-fast",
        "knowledge": "2025-07",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 8000,
          "input": 7000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B-Thinking-2507-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.125,
          "input": 0.1,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "knowledge": "2025-12",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-30B-A3B-Instruct-2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-32B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.125,
          "input": 0.1,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "Qwen/Qwen3-32B",
        "knowledge": "2025-12",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Embedding-8B": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "Qwen/Qwen3-Embedding-8B",
        "knowledge": "2025-10",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Embedding-8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-10",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "Qwen/Qwen3-Next-80B-A3B-Thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "cache_write": 0.18,
          "input": 0.15,
          "output": 1.2,
          "reasoning": 1.2
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "Qwen/Qwen3-Next-80B-A3B-Thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-12",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next-80B-A3B-Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Next-80B-A3B-Thinking-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "cache_write": 0.1875,
          "input": 0.15,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "Qwen/Qwen3-Next-80B-A3B-Thinking-fast",
        "knowledge": "2025-07",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 8000,
          "input": 7000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next-80B-A3B-Thinking-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.75,
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "Qwen/Qwen3.5-397B-A17B",
        "knowledge": "2025-07",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 262144,
          "input": 250000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.75,
          "input": 0.6,
          "output": 3.6
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "Qwen/Qwen3.5-397B-A17B-fast",
        "knowledge": "2025-07",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 8000,
          "input": 7000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-15",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 0.45,
          "reasoning": 0.45
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-11",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 163000,
          "input": 160000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-20",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "cache_write": 0.5,
          "input": 0.4,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "deepseek-ai/DeepSeek-V3.2-fast",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 8000,
          "input": 7000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-01-27",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.75,
          "output": 3.5
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.125,
          "input": 0.1,
          "output": 0.3
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3-27b-it",
        "knowledge": "2025-10",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 110000,
          "input": 100000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma-3-27b-it",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0.013,
          "cache_write": 0.16,
          "input": 0.13,
          "output": 0.4
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "knowledge": "2025-08",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "input": 0.5,
          "output": 2.5,
          "reasoning": 2.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-06",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-15",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "input": 0.5,
          "output": 2.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5-fast",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-06",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-15",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.75,
          "input": 0.6,
          "output": 1.8
        },
        "description": "Flagship Nemotron model for high-throughput reasoning and complex agents",
        "family": "nemotron",
        "id": "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1",
        "knowledge": "2024-12",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 120000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.1-Nemotron-Ultra-253B-v1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B": {
        "attachment": false,
        "cost": {
          "cache_read": 0.006,
          "cache_write": 0.075,
          "input": 0.06,
          "output": 0.24
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B",
        "knowledge": "2025-05",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 32000,
          "input": 30000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron-3-Nano-30B-A3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/Nemotron-3-Nano-Omni": {
        "attachment": false,
        "cost": {
          "cache_read": 0.006,
          "cache_write": 0.075,
          "input": 0.06,
          "output": 0.24
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nvidia/Nemotron-3-Nano-Omni",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 65536,
          "input": 60000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron-3-Nano-Omni",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "knowledge": "2026-02",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron-3-Super-120B-A12B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "cache_write": 0.18,
          "input": 0.15,
          "output": 0.6,
          "reasoning": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "openai/gpt-oss-120b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-09",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "input": 124000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.125,
          "input": 0.1,
          "output": 0.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "openai/gpt-oss-120b-fast",
        "knowledge": "2025-06",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 8000,
          "input": 7000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b-fast",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1,
          "input": 1,
          "output": 3.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01",
        "last_updated": "2026-03-10",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-01",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 432000,
          "output": 432000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Nebius Token Factory",
    "npm": "@ai-sdk/openai-compatible"
  },
  "neon": {
    "api": "${NEON_AI_GATEWAY_BASE_URL}/ai-gateway/mlflow/v1",
    "doc": "https://neon.com/docs",
    "env": [
      "NEON_AI_GATEWAY_BASE_URL",
      "NEON_AI_GATEWAY_TOKEN"
    ],
    "id": "neon",
    "models": {
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2-5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "gemini-2-5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2-5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2-5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3-1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-1-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3-1-pro",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "gemini-3-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "gemini-3-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5-1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "gpt-5-2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "gpt-5-4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.15,
                "input": 1.5,
                "output": 9
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt-mini",
        "id": "gpt-5-4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "gpt-5-4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "gpt-5-5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.072,
          "output": 0.28
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Neon",
    "npm": "@ai-sdk/openai-compatible"
  },
  "neuralwatt": {
    "api": "https://api.neuralwatt.com/v1",
    "doc": "https://portal.neuralwatt.com/docs",
    "env": [
      "NEURALWATT_API_KEY"
    ],
    "id": "neuralwatt",
    "models": {
      "Qwen/Qwen3.5-397B-A17B-FP8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1725,
          "input": 0.69,
          "output": 4.14
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-01",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0725,
          "input": 0.29,
          "output": 1.15
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "Qwen/Qwen3.6-35B-A3B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 131056,
          "output": 131056
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3625,
          "input": 1.45,
          "output": 4.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.2",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 1048560,
          "output": 1048560
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3625,
          "input": 1.45,
          "output": 4.5
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-5.2-fast",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 1048560,
          "output": 1048560
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2-flex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.18125,
          "input": 0.725,
          "output": 2.25
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.2-flex",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 1048560,
          "output": 1048560
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Flex",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2-short": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3625,
          "input": 1.45,
          "output": 4.5
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.2-short",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 199984,
          "output": 199984
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Short",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2-short-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3625,
          "input": 1.45,
          "output": 4.5
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-5.2-short-fast",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 199984,
          "output": 199984
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Short Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2-short-fast-flex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.18125,
          "input": 0.725,
          "output": 2.25
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-5.2-short-fast-flex",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 199984,
          "output": 199984
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Short Fast Flex",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2-short-flex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.18125,
          "input": 0.725,
          "output": 2.25
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.2-short-flex",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 199984,
          "output": 199984
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Short Flex",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-17",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 0.52,
          "output": 2.59
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5-fast",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1725,
          "input": 0.69,
          "output": 3.22
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6-fast",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6-flex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08625,
          "input": 0.345,
          "output": 1.61
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6-flex",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 Flex",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code-flex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11875,
          "input": 0.475,
          "output": 2
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code-flex",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code Flex",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 0.52,
          "output": 2.59
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1725,
          "input": 0.69,
          "output": 3.22
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.7-Code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2375,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.7-Code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "qwen3.5-397b-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1725,
          "input": 0.69,
          "output": 4.14
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "qwen3.5-397b-fast",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 262128,
          "output": 262128
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-35b-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0725,
          "input": 0.29,
          "output": 1.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "qwen3.6",
        "id": "qwen3.6-35b-fast",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 131056,
          "output": 131056
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B Fast",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Neuralwatt",
    "npm": "@ai-sdk/openai-compatible"
  },
  "nova": {
    "api": "https://api.nova.amazon.com/v1",
    "doc": "https://nova.amazon.com/dev/documentation",
    "env": [
      "NOVA_API_KEY"
    ],
    "id": "nova",
    "models": {
      "nova-2-lite-v1": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0,
          "reasoning": 0
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "nova-lite",
        "id": "nova-2-lite-v1",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova 2 Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "nova-2-pro-v1": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0,
          "reasoning": 0
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "nova-pro",
        "id": "nova-2-pro-v1",
        "last_updated": "2026-01-03",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova 2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-03",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Nova",
    "npm": "@ai-sdk/openai-compatible"
  },
  "novita-ai": {
    "api": "https://api.novita.ai/openai",
    "doc": "https://novita.ai/docs/guides/introduction",
    "env": [
      "NOVITA_API_KEY"
    ],
    "id": "novita-ai",
    "models": {
      "baichuan/baichuan-m2-32b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.07
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "baichuan",
        "id": "baichuan/baichuan-m2-32b",
        "knowledge": "2024-12",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "baichuan-m2-32b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-13",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "baidu/ernie-4.5-21B-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "ernie",
        "id": "baidu/ernie-4.5-21B-a3b",
        "knowledge": "2025-03",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 120000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 21B A3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-21B-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "ernie",
        "id": "baidu/ernie-4.5-21B-a3b-thinking",
        "knowledge": "2025-03",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE-4.5-21B-A3B-Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": false
      },
      "baidu/ernie-4.5-300b-a47b-paddle": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.1
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "baidu/ernie-4.5-300b-a47b-paddle",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 123000,
          "output": 12000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 300B A47B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "baidu/ernie-4.5-vl-28b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.56
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "baidu/ernie-4.5-vl-28b-a3b",
        "last_updated": "2026-06-14",
        "limit": {
          "context": 30000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 VL 28B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-vl-28b-a3b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.39,
          "output": 0.39
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "baidu/ernie-4.5-vl-28b-a3b-thinking",
        "last_updated": "2025-11-26",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE-4.5-VL-28B-A3B-Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-vl-424b-a47b": {
        "attachment": true,
        "cost": {
          "input": 0.42,
          "output": 1.25
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "baidu/ernie-4.5-vl-424b-a47b",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 123000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 VL 424B A47B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-30",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-ocr": {
        "attachment": true,
        "cost": {
          "input": 0.03,
          "output": 0.03
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "id": "deepseek/deepseek-ocr",
        "last_updated": "2025-10-24",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-OCR",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-ocr-2": {
        "attachment": true,
        "cost": {
          "input": 0.03,
          "output": 0.03
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "id": "deepseek/deepseek-ocr-2",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek/deepseek-ocr-2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-27",
        "tool_call": false
      },
      "deepseek/deepseek-prover-v2-671b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.5
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "deepseek/deepseek-prover-v2-671b",
        "last_updated": "2025-04-30",
        "limit": {
          "context": 160000,
          "output": 160000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek Prover V2 671B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-30",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "cache_read": 0.35,
          "input": 0.7,
          "output": 2.5
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-0528",
        "knowledge": "2024-07",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-0528-qwen3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.09
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek/deepseek-r1-0528-qwen3-8b",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528 Qwen3 8B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-29",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 0.8
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-distill-llama-70b",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill LLama 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1-distill-qwen-14b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-distill-qwen-14b",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 14B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1-distill-qwen-32b": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-distill-qwen-32b",
        "last_updated": "2025-01-20",
        "limit": {
          "context": 64000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 Distill Qwen 32B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.5
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek/deepseek-r1-turbo",
        "last_updated": "2025-03-05",
        "limit": {
          "context": 64000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 (Turbo)\t",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-05",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3-0324": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.27,
          "output": 1.12
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3-0324",
        "knowledge": "2024-07",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.3
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "id": "deepseek/deepseek-v3-turbo",
        "last_updated": "2025-03-05",
        "limit": {
          "context": 64000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 (Turbo)\t",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-05",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1-terminus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1-terminus",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek V3.1 Terminus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1345,
          "input": 0.269,
          "output": 0.4
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-exp": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.41
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-exp",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek V3.2 Exp",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 393216
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 1.6,
          "output": 3.2
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 393216
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3-12b-it": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.1
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3-12b-it",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.119,
          "output": 0.2
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3-27b-it",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 98304,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-25",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gryphe/mythomax-l2-13b": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.09
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "gryphe/mythomax-l2-13b",
        "last_updated": "2024-04-25",
        "limit": {
          "context": 4096,
          "output": 3200
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mythomax L2 13B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-25",
        "temperature": true,
        "tool_call": false
      },
      "inclusionai/ling-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "ling",
        "id": "inclusionai/ling-2.6-1t",
        "last_updated": "2026-06-29",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling-2.6-1T",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "ling",
        "id": "inclusionai/ling-2.6-flash",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling-2.6-flash",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ring-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "ring",
        "id": "inclusionai/ring-2.6-1t",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ring-2.6-1T",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kwaipilot/kat-coder-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "kwaipilot/kat-coder-pro",
        "last_updated": "2026-01-05",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kat Coder Pro",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.51,
          "output": 0.74
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3-70b-instruct",
        "last_updated": "2024-04-25",
        "limit": {
          "context": 8192,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3-8b-instruct",
        "last_updated": "2024-04-25",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-25",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.1-8b-instruct",
        "last_updated": "2024-07-24",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-24",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.2-3b-instruct",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 32768,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.135,
          "output": 0.4
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-07",
        "limit": {
          "context": 131072,
          "output": 120000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-07",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-4-maverick-17b-128e-instruct-fp8": {
        "attachment": true,
        "cost": {
          "input": 0.27,
          "output": 0.85
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "id": "meta-llama/llama-4-maverick-17b-128e-instruct-fp8",
        "last_updated": "2025-04-06",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-06",
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-4-scout-17b-16e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.59
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "id": "meta-llama/llama-4-scout-17b-16e-instruct",
        "last_updated": "2025-04-06",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-06",
        "temperature": true,
        "tool_call": false
      },
      "microsoft/wizardlm-2-8x22b": {
        "attachment": false,
        "cost": {
          "input": 0.62,
          "output": 0.62
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "microsoft/wizardlm-2-8x22b",
        "last_updated": "2024-04-24",
        "limit": {
          "context": 65535,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Wizardlm 2 8x22B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-24",
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-10-27",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax M2.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax-m2.5",
        "id": "minimax/minimax-m2.5-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5 Highspeed",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax-m2.7",
        "id": "minimax/minimax-m2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax/minimax-m2.7-highspeed",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimaxai/minimax-m1-80k": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 2.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimaxai/minimax-m1-80k",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1000000,
          "output": 40000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.17
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistralai/mistral-nemo",
        "last_updated": "2024-07-30",
        "limit": {
          "context": 60288,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "moonshotai/kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-0905",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.57,
          "output": 2.3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "moonshotai/kimi-k2-instruct",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-29",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.8,
          "output": 3.4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nousresearch/hermes-2-pro-llama-3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.14
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "nousresearch/hermes-2-pro-llama-3-8b",
        "last_updated": "2024-06-27",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 2 Pro Llama 3 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-oss-120b": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": true,
        "cost": {
          "input": 0.04,
          "output": 0.15
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI: GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "paddlepaddle/paddleocr-vl": {
        "attachment": true,
        "cost": {
          "input": 0.02,
          "output": 0.02
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "paddlepaddle/paddleocr-vl",
        "last_updated": "2025-10-22",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "PaddleOCR-VL",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-22",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen-2.5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.38,
          "output": 0.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen-2.5-72b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2024-10-15",
        "limit": {
          "context": 32000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-mt-plus": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.75
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "id": "qwen/qwen-mt-plus",
        "last_updated": "2025-09-03",
        "limit": {
          "context": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen MT Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-03",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen2.5-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.07
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen2.5-7b-instruct",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen2.5-vl-72b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 0.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen2.5-vl-72b-instruct",
        "last_updated": "2025-03-25",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 VL 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-25",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-235b-a22b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-235b-a22b-fp8",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 40960,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.58
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 3
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-thinking-2507",
        "knowledge": "2025-04",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22b Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.45
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-30b-a3b-fp8",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 40960,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-32b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.45
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-32b-fp8",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 40960,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-4b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.03
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-4b-fp8",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 128000,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 4B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-8b-fp8": {
        "attachment": false,
        "cost": {
          "input": 0.035,
          "output": 0.138
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-8b-fp8",
        "last_updated": "2025-04-29",
        "limit": {
          "context": 128000,
          "output": 20000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 8B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-29",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.27
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-30b-a3b-instruct",
        "last_updated": "2025-10-09",
        "limit": {
          "context": 160000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 30b A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.38,
          "output": 1.55
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-next",
        "last_updated": "2026-02-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 2.11,
          "output": 8.45
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen/qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3-next-80b-a3b-instruct",
        "last_updated": "2025-09-10",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-next-80b-a3b-thinking",
        "last_updated": "2025-09-10",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-omni-30b-a3b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "input_audio": 2.2,
          "output": 0.97,
          "output_audio": 1.788
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-omni-30b-a3b-instruct",
        "knowledge": "2024-04",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "video",
            "audio",
            "image"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Qwen3 Omni 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-omni-30b-a3b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "input_audio": 2.2,
          "output": 0.97,
          "output_audio": 1.788
        },
        "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
        "id": "qwen/qwen3-omni-30b-a3b-thinking",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "audio",
            "video",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Omni 30B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-235b-a22b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-235b-a22b-instruct",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-235b-a22b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.98,
          "output": 3.95
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-235b-a22b-thinking",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-24",
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-vl-30b-a3b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.7
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-30b-a3b-instruct",
        "last_updated": "2025-10-11",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "video",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen/qwen3-vl-30b-a3b-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-30b-a3b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-30b-a3b-thinking",
        "last_updated": "2025-10-11",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen/qwen3-vl-30b-a3b-thinking",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-8b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "output": 0.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3-vl-8b-instruct",
        "last_updated": "2025-10-17",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen/qwen3-vl-8b-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-122b-a10b",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-27b",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-35b-a3b",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-397b-a17b",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 262144,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 1.5625,
          "input": 1.25,
          "output": 3.75
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen/qwen3.7-max",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7-Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "sao10K/L3-8B-stheno-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "sao10K/L3-8B-stheno-v3.2",
        "last_updated": "2024-11-29",
        "limit": {
          "context": 8192,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "L3 8B Stheno V3.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-29",
        "temperature": true,
        "tool_call": true
      },
      "sao10K/l3-70b-euryale-v2.1": {
        "attachment": false,
        "cost": {
          "input": 1.48,
          "output": 1.48
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "sao10K/l3-70b-euryale-v2.1",
        "last_updated": "2024-06-18",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "L3 70B Euryale V2.1\t",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-18",
        "temperature": true,
        "tool_call": true
      },
      "sao10K/l3-8b-lunaris": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.05
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "sao10K/l3-8b-lunaris",
        "last_updated": "2024-11-28",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sao10k L3 8B Lunaris\t",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "sao10K/l31-70b-euryale-v2.2": {
        "attachment": false,
        "cost": {
          "input": 1.48,
          "output": 1.48
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "sao10K/l31-70b-euryale-v2.2",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "L31 70B Euryale V2.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-19",
        "temperature": true,
        "tool_call": true
      },
      "xiaomimimo/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomimimo/mimo-v2-flash",
        "knowledge": "2024-12",
        "last_updated": "2025-12-19",
        "limit": {
          "context": 262144,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "XiaomiMiMo/MiMo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xiaomimimo/mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.4,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 2,
          "output": 6,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "xiaomimimo/mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xiaomimimo/mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0043,
          "context_over_200k": {
            "cache_read": 0.0043,
            "input": 0.522,
            "output": 1.044
          },
          "input": 0.522,
          "output": 1.044,
          "tiers": [
            {
              "cache_read": 0.0043,
              "input": 0.522,
              "output": 1.044,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "xiaomimimo/mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/autoglm-phone-9b-multilingual": {
        "attachment": true,
        "cost": {
          "input": 0.035,
          "output": 0.138
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "zai-org/autoglm-phone-9b-multilingual",
        "last_updated": "2025-12-10",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "AutoGLM-Phone-9B-Multilingual",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-10",
        "temperature": true,
        "tool_call": false
      },
      "zai-org/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.13,
          "output": 0.85
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "zai-org/glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-10-13",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-10-13",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.5v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glmv",
        "id": "zai-org/glm-4.5v",
        "knowledge": "2025-04",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "video",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.55,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.6v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.055,
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glmv",
        "id": "zai-org/glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "video",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "zai-org/glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.38,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "NovitaAI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "nvidia": {
    "api": "https://integrate.api.nvidia.com/v1",
    "doc": "https://docs.api.nvidia.com/nim/",
    "env": [
      "NVIDIA_API_KEY"
    ],
    "id": "nvidia",
    "models": {
      "abacusai/dracarys-llama-3_1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "abacusai/dracarys-llama-3_1-70b-instruct",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "dracarys-llama-3.1-70b-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-11",
        "temperature": true,
        "tool_call": true
      },
      "baai/bge-m3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "bge",
        "id": "baai/bge-m3",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 8192,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE M3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-01-30",
        "temperature": false,
        "tool_call": false
      },
      "black-forest-labs/flux.1-dev": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "black-forest-labs/flux.1-dev",
        "knowledge": "2024-08",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 4096,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1-dev",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "temperature": true,
        "tool_call": false
      },
      "black-forest-labs/flux_1-kontext-dev": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "black-forest-labs/flux_1-kontext-dev",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1-Kontext-dev",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-12",
        "temperature": false,
        "tool_call": false
      },
      "black-forest-labs/flux_1-schnell": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "black-forest-labs/flux_1-schnell",
        "knowledge": "2024-07",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 77,
          "input": 77,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1-schnell",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-01",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "black-forest-labs/flux_2-klein-4b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "black-forest-labs/flux_2-klein-4b",
        "knowledge": "2025-06",
        "last_updated": "2026-01-31",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 Klein 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-14",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seed-oss-36b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "seed",
        "id": "bytedance/seed-oss-36b-instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance-Seed/Seed-OSS-36B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-ai/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 393216
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 393216
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-2-2b-it": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-2-2b-it",
        "last_updated": "2024-07-16",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 2 2b It",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3n-e2b-it": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3n-e2b-it",
        "knowledge": "2024-06",
        "last_updated": "2025-06-12",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3n E2b It",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3n-e4b-it": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-3n-e4b-it",
        "knowledge": "2024-06",
        "last_updated": "2025-06-03",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3n E4b It",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "knowledge": "2025-01",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma-4-31B-IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "google/google-paligemma": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Gemini multimodal model for text, image, audio, video, and document tasks",
        "id": "google/google-paligemma",
        "last_updated": "2024-08-26",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "paligemma",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-05-14",
        "temperature": true,
        "tool_call": false
      },
      "meta/esm2-650m": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta/esm2-650m",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "esm2-650m",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-29",
        "temperature": true,
        "tool_call": false
      },
      "meta/esmfold": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta/esmfold",
        "last_updated": "2025-06-12",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "esmfold",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-03-15",
        "temperature": true,
        "tool_call": false
      },
      "meta/llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta/llama-3.1-70b-instruct",
        "last_updated": "2024-07-16",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70b Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.1-8b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2025-01-01",
        "limit": {
          "context": 16000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "id": "meta/llama-3.2-11b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11b Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-1b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta/llama-3.2-1b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 1b Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.2-3b-instruct",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 32768,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "meta/llama-3.2-90b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta/llama-3.2-90b-vision-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.2-90B-Vision-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta/llama-3.3-70b-instruct",
        "last_updated": "2024-11-26",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70b Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-4-maverick-17b-128e-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "id": "meta/llama-4-maverick-17b-128e-instruct",
        "knowledge": "2024-02",
        "last_updated": "2025-04-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick 17b 128e Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-guard-4-12b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "meta/llama-guard-4-12b",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama Guard 4 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": false
      },
      "microsoft/phi-4-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/phi-4-mini-instruct",
        "knowledge": "2024-12",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-Mini",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "microsoft/phi-4-multimodal-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "microsoft/phi-4-multimodal-instruct",
        "last_updated": "2025-07-26",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi 4 Multimodal",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-26",
        "structured_output": false,
        "tool_call": false
      },
      "minimaxai/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimaxai/minimax-m2.7",
        "last_updated": "2026-04-11",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimaxai/minimax-m3": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimaxai/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/magistral-small-2506": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "id": "mistralai/magistral-small-2506",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Small 2506",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-7b-instruct-v03": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mistral-7b-instruct-v03",
        "last_updated": "2025-04-01",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral-7B-Instruct-v0.3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-3-675b-instruct-2512": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistralai/mistral-large-3-675b-instruct-2512",
        "knowledge": "2025-01",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3 675B Instruct 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3-instruct": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/mistral-medium-3-instruct",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 131072,
          "input": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-25",
        "structured_output": false,
        "tool_call": false
      },
      "mistralai/mistral-nemotron": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "nemotron",
        "id": "mistralai/mistral-nemotron",
        "last_updated": "2025-06-12",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "mistral-nemotron",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-11",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-small-4-119b-2603": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistralai/mistral-small-4-119b-2603",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "mistral-small-4-119b-2603",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mixtral-8x22b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mixtral-8x22b-instruct",
        "last_updated": "2024-04-17",
        "limit": {
          "context": 65536,
          "output": 13108
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mixtral 8x22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-17",
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mixtral-8x7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistralai/mixtral-8x7b-instruct",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral: Mixtral 8x7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-12-10",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-instruct-0905": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-instruct-0905",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/active-speaker-detection": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "id": "nvidia/active-speaker-detection",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Active Speaker Detection",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/bevformer": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "id": "nvidia/bevformer",
        "last_updated": "2025-07-20",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "bevformer",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-18",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/cosmos-predict1-5b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "nvidia/cosmos-predict1-5b",
        "last_updated": "2025-03-18",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "video"
          ]
        },
        "name": "cosmos-predict1-5b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-18",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/cosmos-transfer1-7b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "nvidia/cosmos-transfer1-7b",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "video"
          ]
        },
        "name": "cosmos-transfer1-7b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-13",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/cosmos-transfer2_5-2b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "nvidia/cosmos-transfer2_5-2b",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "video"
          ]
        },
        "name": "cosmos-transfer2.5-2b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-26",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/gliner-pii": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "id": "nvidia/gliner-pii",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gliner-pii",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/llama-3_1-nemotron-safety-guard-8b-v3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "nemotron",
        "id": "nvidia/llama-3_1-nemotron-safety-guard-8b-v3",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "llama-3.1-nemotron-safety-guard-8b-v3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-28",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/llama-3_2-nemoretriever-300m-embed-v1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "id": "nvidia/llama-3_2-nemoretriever-300m-embed-v1",
        "last_updated": "2025-07-24",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "llama-3_2-nemoretriever-300m-embed-v1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-24",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/llama-nemotron-embed-vl-1b-v2": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "nemotron",
        "id": "nvidia/llama-nemotron-embed-vl-1b-v2",
        "last_updated": "2026-02-10",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "llama-nemotron-embed-vl-1b-v2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-10",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/llama-nemotron-rerank-vl-1b-v2": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "nemotron",
        "id": "nvidia/llama-nemotron-rerank-vl-1b-v2",
        "last_updated": "2026-03-31",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "llama-nemotron-rerank-vl-1b-v2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-31",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/magpie-tts-zeroshot": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "id": "nvidia/magpie-tts-zeroshot",
        "last_updated": "2025-06-12",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "magpie-tts-zeroshot",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-22",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/nemotron-3-content-safety": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-content-safety",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-3-content-safety",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/nemotron-3-nano-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-30b-a3b",
        "knowledge": "2024-09",
        "last_updated": "2024-12",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-3-nano-30b-a3b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2024-12",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning",
        "last_updated": "2026-04-28",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Nano Omni",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": -1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "knowledge": "2024-04",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-ultra-550b-a55b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.5,
          "output": 2.5
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-ultra-550b-a55b",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra 550B A55B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-content-safety-reasoning-4b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "nemotron",
        "id": "nvidia/nemotron-content-safety-reasoning-4b",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-content-safety-reasoning-4b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-22",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/nemotron-mini-4b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-mini-4b-instruct",
        "last_updated": "2024-08-26",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-mini-4b-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-21",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-voicechat": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "family": "nemotron",
        "id": "nvidia/nemotron-voicechat",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-voicechat",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nv-embed-v1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "id": "nvidia/nv-embed-v1",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nv-embed-v1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-07",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/nv-embedcode-7b-v1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "id": "nvidia/nv-embedcode-7b-v1",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 32768,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nv-embedcode-7b-v1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-17",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/nvidia-nemotron-nano-9b-v2": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia/nvidia-nemotron-nano-9b-v2",
        "knowledge": "2024-09",
        "last_updated": "2025-08-18",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nvidia-nemotron-nano-9b-v2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-18",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/rerank-qa-mistral-4b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "id": "nvidia/rerank-qa-mistral-4b",
        "last_updated": "2025-01-17",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "rerank-qa-mistral-4b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-03-17",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/riva-translate-4b-instruct-v1_1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Translation model for multilingual conversion, localization, and cross-language workflows",
        "id": "nvidia/riva-translate-4b-instruct-v1_1",
        "last_updated": "2025-12-12",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "riva-translate-4b-instruct-v1_1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-12",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/sparsedrive": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "id": "nvidia/sparsedrive",
        "last_updated": "2025-07-20",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "sparsedrive",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-18",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/streampetr": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "id": "nvidia/streampetr",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "streampetr",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-11-13",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/studiovoice": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "id": "nvidia/studiovoice",
        "last_updated": "2025-06-13",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "studiovoice",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-03",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/synthetic-video-detector": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "nvidia/synthetic-video-detector",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "synthetic-video-detector",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-16",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/usdcode": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "id": "nvidia/usdcode",
        "last_updated": "2026-01-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "usdcode",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-01",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/usdvalidate": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "id": "nvidia/usdvalidate",
        "last_updated": "2025-01-08",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "usdvalidate",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-24",
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "knowledge": "2025-08",
        "last_updated": "2025-08-14",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS-120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/whisper-large-v3": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "openai/whisper-large-v3",
        "knowledge": "2023-09",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper Large v3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-09-01",
        "temperature": false,
        "tool_call": false
      },
      "qwen/qwen-image": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen/qwen-image",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen-image-edit": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen/qwen-image-edit",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen Image Edit",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen2.5-coder-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen2.5-coder-32b-instruct",
        "last_updated": "2024-11-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 Coder 32b Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-480b-a35b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 66536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-next-80b-a3b-instruct",
        "knowledge": "2024-12",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next-80B-A3B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-122b-a10b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-397b-a17b",
        "knowledge": "2026-01",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "sarvamai/sarvam-m": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work",
        "id": "sarvamai/sarvam-m",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "sarvam-m",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun-ai/step-3.5-flash",
        "last_updated": "2026-02-02",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-02",
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/step-3.7-flash": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun-ai/step-3.7-flash",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": true
      },
      "upstage/solar-10_7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "upstage/solar-10_7b-instruct",
        "last_updated": "2025-04-10",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "solar-10.7b-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-06-05",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.1": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "z-ai/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Nvidia",
    "npm": "@ai-sdk/openai-compatible"
  },
  "ollama-cloud": {
    "api": "https://ollama.com/v1",
    "doc": "https://docs.ollama.com/cloud",
    "env": [
      "OLLAMA_API_KEY"
    ],
    "id": "ollama-cloud",
    "models": {
      "cogito-2.1:671b": {
        "attachment": false,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "cogito",
        "id": "cogito-2.1:671b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 163840,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "cogito-2.1:671b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-11-19",
        "status": "deprecated",
        "tool_call": true
      },
      "deepseek-v3.1:671b": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.1:671b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-v3.1:671b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-21",
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-v3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-06-15",
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 1048576
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-v4-flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 1048576
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-v4-pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "tool_call": true
      },
      "devstral-2:123b": {
        "attachment": false,
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "devstral-2:123b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "devstral-2:123b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "tool_call": true
      },
      "devstral-small-2:24b": {
        "attachment": true,
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "devstral-small-2:24b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "devstral-small-2:24b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-04-08",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-3-flash-preview",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-17",
        "tool_call": true
      },
      "gemma3:12b": {
        "attachment": true,
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma3:12b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemma3:12b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "tool_call": false
      },
      "gemma3:27b": {
        "attachment": true,
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma3:27b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemma3:27b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-27",
        "tool_call": false
      },
      "gemma3:4b": {
        "attachment": true,
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma3:4b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemma3:4b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "tool_call": false
      },
      "gemma4:31b": {
        "attachment": true,
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma4:31b",
        "knowledge": "2025-01",
        "last_updated": "2026-04-08",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemma4:31b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "glm",
        "id": "glm-4.6",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "status": "deprecated",
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 976000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss:120b": {
        "attachment": false,
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss:120b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss:120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "tool_call": true
      },
      "gpt-oss:20b": {
        "attachment": false,
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss:20b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss:20b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "knowledge": "2024-08",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2-thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "status": "deprecated",
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-20",
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2.7-code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2:1t": {
        "attachment": false,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-k2",
        "id": "kimi-k2:1t",
        "knowledge": "2024-10",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2:1t",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "status": "deprecated",
        "tool_call": true
      },
      "minimax-m2": {
        "attachment": false,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax",
        "id": "minimax-m2",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "minimax-m2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-23",
        "status": "deprecated",
        "tool_call": true
      },
      "minimax-m2.1": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.1",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "minimax-m2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.5",
        "knowledge": "2025-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "minimax-m2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "tool_call": true
      },
      "minimax-m2.7": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "minimax-m2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "tool_call": true
      },
      "minimax-m3": {
        "attachment": true,
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax-m3",
        "id": "minimax-m3",
        "knowledge": "2025-01",
        "last_updated": "2026-05-31",
        "limit": {
          "context": 512000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "minimax-m3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-31",
        "temperature": true,
        "tool_call": true
      },
      "ministral-3:14b": {
        "attachment": true,
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3:14b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ministral-3:14b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "tool_call": true
      },
      "ministral-3:3b": {
        "attachment": true,
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3:3b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ministral-3:3b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-22",
        "tool_call": true
      },
      "ministral-3:8b": {
        "attachment": true,
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "ministral-3:8b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ministral-3:8b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-01",
        "tool_call": true
      },
      "mistral-large-3:675b": {
        "attachment": true,
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large-3:675b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "mistral-large-3:675b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "tool_call": true
      },
      "nemotron-3-nano:30b": {
        "attachment": false,
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nemotron-3-nano:30b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-3-nano:30b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-15",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-super": {
        "attachment": false,
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nemotron-3-super",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-3-super",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-ultra": {
        "attachment": false,
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nemotron-3-ultra",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "nemotron-3-ultra",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-next": {
        "attachment": false,
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-next",
        "last_updated": "2026-02-08",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-coder-next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-02",
        "tool_call": true
      },
      "qwen3-coder:480b": {
        "attachment": false,
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder:480b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-coder:480b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-22",
        "tool_call": true
      },
      "qwen3-next:80b": {
        "attachment": false,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen",
        "id": "qwen3-next:80b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-next:80b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-15",
        "status": "deprecated",
        "tool_call": true
      },
      "qwen3-vl:235b": {
        "attachment": true,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen",
        "id": "qwen3-vl:235b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-vl:235b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-22",
        "status": "deprecated",
        "tool_call": true
      },
      "qwen3-vl:235b-instruct": {
        "attachment": true,
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen",
        "id": "qwen3-vl:235b-instruct",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-vl:235b-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-22",
        "status": "deprecated",
        "tool_call": true
      },
      "qwen3.5:397b": {
        "attachment": true,
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5:397b",
        "interleaved": {
          "field": "reasoning_details"
        },
        "last_updated": "2026-02-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3.5:397b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-15",
        "tool_call": true
      },
      "rnj-1:8b": {
        "attachment": false,
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "rnj",
        "id": "rnj-1:8b",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "rnj-1:8b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-06",
        "tool_call": true
      }
    },
    "name": "Ollama Cloud",
    "npm": "@ai-sdk/openai-compatible"
  },
  "openai": {
    "doc": "https://platform.openai.com/docs/models",
    "env": [
      "OPENAI_API_KEY"
    ],
    "id": "openai",
    "models": {
      "chatgpt-image-latest": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "chatgpt-image-latest",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "input": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "chatgpt-image-latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": false,
        "tool_call": false
      },
      "gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-3.5-turbo",
        "knowledge": "2021-09-01",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gpt-4": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-2024-05-13": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 15
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o-2024-05-13",
        "knowledge": "2023-09",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-05-13)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-2024-08-06": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o-2024-08-06",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-08-06)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4o-2024-11-20",
        "knowledge": "2023-09",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "gpt-5-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations",
        "family": "gpt-codex",
        "id": "gpt-5.1-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Codex GPT for repository edits, code review, and practical software agents",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "gpt-5.2-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows",
        "family": "gpt-pro",
        "id": "gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "gpt-5.3-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex-spark": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex-spark",
        "id": "gpt-5.3-codex-spark",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 128000,
          "input": 100000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex Spark",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.15,
                "input": 1.5,
                "output": 9
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt-pro",
        "id": "gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-image-1": {
        "attachment": true,
        "description": "OpenAI image model for production generation, edits, and brand-safe visual workflows",
        "family": "gpt-image",
        "id": "gpt-image-1",
        "last_updated": "2025-04-24",
        "limit": {
          "context": 0,
          "input": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "gpt-image-1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-24",
        "temperature": false,
        "tool_call": false
      },
      "gpt-image-1-mini": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "gpt-image-1-mini",
        "last_updated": "2025-09-26",
        "limit": {
          "context": 0,
          "input": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "gpt-image-1-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-26",
        "temperature": false,
        "tool_call": false
      },
      "gpt-image-1.5": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "gpt-image-1.5",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 0,
          "input": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "gpt-image-1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-25",
        "temperature": false,
        "tool_call": false
      },
      "gpt-image-2": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 30
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "gpt-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 0,
          "input": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "gpt-image-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o1-pro": {
        "attachment": true,
        "cost": {
          "input": 150,
          "output": 600
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "o1-pro",
        "knowledge": "2023-09",
        "last_updated": "2025-03-19",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-03-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o3-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 2.5,
          "input": 10,
          "output": 40
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o",
        "id": "o3-deep-research",
        "knowledge": "2024-05",
        "last_updated": "2024-06-26",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2024-06-26",
        "temperature": false,
        "tool_call": true
      },
      "o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o3-pro": {
        "attachment": true,
        "cost": {
          "input": 20,
          "output": 80
        },
        "description": "High-effort o3 tier for difficult technical reasoning and careful answers",
        "family": "o-pro",
        "id": "o3-pro",
        "knowledge": "2024-05",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "o4-mini-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o-mini",
        "id": "o4-mini-deep-research",
        "knowledge": "2024-05",
        "last_updated": "2024-06-26",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2024-06-26",
        "temperature": false,
        "tool_call": true
      },
      "text-embedding-3-large": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-3-large",
        "knowledge": "2024-01",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8191,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": false,
        "tool_call": false
      },
      "text-embedding-3-small": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-3-small",
        "knowledge": "2024-01",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8191,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-small",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": false,
        "tool_call": false
      },
      "text-embedding-ada-002": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "text-embedding-ada-002",
        "knowledge": "2022-12",
        "last_updated": "2022-12-15",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-ada-002",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2022-12-15",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "OpenAI",
    "npm": "@ai-sdk/openai"
  },
  "opencode": {
    "api": "https://opencode.ai/zen/v1",
    "doc": "https://opencode.ai/docs/zen",
    "env": [
      "OPENCODE_API_KEY"
    ],
    "id": "opencode",
    "models": {
      "big-pickle": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "big-pickle",
        "id": "big-pickle",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2025-10-17",
        "limit": {
          "context": 200000,
          "input": 160000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Big Pickle",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-3-5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-haiku",
        "id": "claude-3-5-haiku",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": false,
        "release_date": "2024-10-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "interleaved": true,
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "interleaved": true,
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek-flash-free",
        "id": "deepseek-v4-flash-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.145,
          "input": 1.74,
          "output": 3.84
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gemini-pro",
        "id": "gemini-3-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "input_audio": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/google"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "glm-free",
        "id": "glm-4.7-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7 Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-5-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "glm-free",
        "id": "glm-5-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5 Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.107,
          "input": 1.07,
          "output": 8.5
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.107,
          "input": 1.07,
          "output": 8.5
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Codex",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.107,
          "input": 1.07,
          "output": 8.5
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.107,
          "input": 1.07,
          "output": 8.5
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Mini",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex-spark": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex-spark",
        "id": "gpt-5.3-codex-spark",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 128000,
          "input": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex Spark",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Nano",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 30,
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 30,
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 2
        },
        "description": "Grok coding model for agentic engineering, edits, and codebase workflows",
        "family": "grok-build",
        "id": "grok-build-0.1",
        "last_updated": "2026-05-20",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "grok",
        "id": "grok-code",
        "last_updated": "2025-08-20",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Code Fast 1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-20",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "hy3-preview-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "hy3-free",
        "id": "hy3-preview-free",
        "knowledge": "2025-06",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3 preview Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-20",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.4,
          "input": 0.4,
          "output": 2.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-k2",
        "id": "kimi-k2",
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-05",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.4,
          "input": 0.4,
          "output": 2.5
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-thinking",
        "id": "kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-05",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-free",
        "id": "kimi-k2.5-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5 Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "ling-2.6-flash-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "ling-flash-free",
        "id": "ling-2.6-flash-free",
        "knowledge": "2025-06",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262100,
          "output": 32800
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling 2.6 Flash Free",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-21",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-flash-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo-flash-free",
        "id": "mimo-v2-flash-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-16",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-omni-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo-omni-free",
        "id": "mimo-v2-omni-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Omni Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo-pro-free",
        "id": "mimo-v2-pro-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1048576,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Pro Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo-v2.5-free",
        "id": "mimo-v2.5-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5 Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax",
        "id": "minimax-m2.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.1-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax-free",
        "id": "minimax-m2.1-free",
        "knowledge": "2025-01",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1 Free",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax-free",
        "id": "minimax-m2.5-free",
        "knowledge": "2025-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5 Free",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax-m3",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax-m3-free",
        "id": "minimax-m3-free",
        "knowledge": "2025-01",
        "last_updated": "2026-05-31",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3 Free",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-31",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-super-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron-free",
        "id": "nemotron-3-super-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-02",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-11",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "nemotron-3-ultra-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron-free",
        "id": "nemotron-3-ultra-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-02",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-04",
        "temperature": true,
        "tool_call": true
      },
      "north-mini-code-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere coding model for practical software engineering and agentic edits",
        "family": "north-free",
        "id": "north-mini-code-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-09-23",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "North Mini Code Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 1.8
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen",
        "id": "qwen3-coder",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.25,
          "input": 0.2,
          "output": 1.2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qwen3.5",
        "id": "qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "input": 0.5,
          "output": 3
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qwen3.6",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen-free",
        "id": "qwen3.6-plus-free",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus Free",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "ring-2.6-1t-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "ring-1t-free",
        "id": "ring-2.6-1t-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-06",
        "last_updated": "2026-05-08",
        "limit": {
          "context": 262000,
          "output": 66000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ring 2.6 1T Free",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-08",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "trinity-large-preview-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "trinity",
        "id": "trinity-large-preview-free",
        "knowledge": "2025-06",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Large Preview",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-28",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "OpenCode Zen",
    "npm": "@ai-sdk/openai-compatible"
  },
  "opencode-go": {
    "api": "https://opencode.ai/zen/go/v1",
    "doc": "https://opencode.ai/docs/zen",
    "env": [
      "OPENCODE_API_KEY"
    ],
    "id": "opencode-go",
    "models": {
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0145,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-27",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-10",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2.7-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo-v2-omni",
        "id": "mimo-v2-omni",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Omni",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo-v2-pro",
        "id": "mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1048576,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo-v2.5",
        "id": "mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0145,
          "input": 1.74,
          "output": 3.48
        },
        "description": "MiMo pro model for strong multimodal reasoning and agent execution",
        "family": "mimo-v2.5-pro",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax-m2.5",
        "id": "minimax-m2.5",
        "knowledge": "2025-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax-m2.7",
        "id": "minimax-m2.7",
        "knowledge": "2025-01",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "context_over_200k": {
            "cache_read": 0.12,
            "input": 0.6,
            "output": 2.4
          },
          "input": 0.3,
          "output": 1.2,
          "tiers": [
            {
              "cache_read": 0.12,
              "input": 0.6,
              "output": 2.4,
              "tier": {
                "size": 512000,
                "type": "context"
              }
            }
          ]
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax-m3",
        "id": "minimax-m3",
        "knowledge": "2025-01",
        "last_updated": "2026-05-31",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-31",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.25,
          "input": 0.2,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen3.5",
        "id": "qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qwen3.6",
        "id": "qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "qwen3.7-max",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "cache_write": 0.5,
          "context_over_200k": {
            "cache_read": 0.12,
            "cache_write": 1.5,
            "input": 1.2,
            "output": 4.8
          },
          "input": 0.4,
          "output": 1.6,
          "tiers": [
            {
              "cache_read": 0.12,
              "cache_write": 1.5,
              "input": 1.2,
              "output": 4.8,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "qwen3.7-plus",
        "id": "qwen3.7-plus",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "OpenCode Go",
    "npm": "@ai-sdk/openai-compatible"
  },
  "openrouter": {
    "api": "https://openrouter.ai/api/v1",
    "doc": "https://openrouter.ai/models",
    "env": [
      "OPENROUTER_API_KEY"
    ],
    "id": "openrouter",
    "models": {
      "ai21/jamba-large-1.7": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "jamba",
        "id": "ai21/jamba-large-1.7",
        "knowledge": "2024-08-31",
        "last_updated": "2025-08-08",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Jamba Large 1.7",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "aion-labs/aion-1.0": {
        "attachment": false,
        "cost": {
          "input": 4,
          "output": 8
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "aion-labs/aion-1.0",
        "last_updated": "2025-02-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion-1.0",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "aion-labs/aion-1.0-mini": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 1.4
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "aion-labs/aion-1.0-mini",
        "last_updated": "2025-02-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion-1.0-Mini",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "aion-labs/aion-2.0": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.8,
          "output": 1.6
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "aion-labs/aion-2.0",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion-2.0",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "aion-labs/aion-rp-llama-3.1-8b": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.6
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "aion-labs/aion-rp-llama-3.1-8b",
        "knowledge": "2023-12-31",
        "last_updated": "2025-02-04",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion-RP 1.0 (8B)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "allenai/olmo-3-32b-think": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "allenai",
        "id": "allenai/olmo-3-32b-think",
        "last_updated": "2025-11-21",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Olmo 3 32B Think",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "amazon/nova-2-lite-v1": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 2.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "nova",
        "id": "amazon/nova-2-lite-v1",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 1000000,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova 2 Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-02",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-lite-v1": {
        "attachment": true,
        "cost": {
          "input": 0.06,
          "output": 0.24
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-lite",
        "id": "amazon/nova-lite-v1",
        "knowledge": "2024-10-31",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 300000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Lite 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-micro-v1": {
        "attachment": false,
        "cost": {
          "input": 0.035,
          "output": 0.14
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-micro",
        "id": "amazon/nova-micro-v1",
        "knowledge": "2024-10-31",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 128000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Micro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-premier-v1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.625,
          "input": 2.5,
          "output": 12.5
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova",
        "id": "amazon/nova-premier-v1",
        "last_updated": "2025-10-31",
        "limit": {
          "context": 1000000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Premier 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-31",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-pro-v1": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 3.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova-pro",
        "id": "amazon/nova-pro-v1",
        "knowledge": "2024-10-31",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 300000,
          "output": 5120
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Pro 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "anthracite-org/magnum-v4-72b": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "anthracite-org/magnum-v4-72b",
        "knowledge": "2024-06-30",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 16384,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magnum v4 72B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "anthropic/claude-3-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude",
        "id": "anthropic/claude-3-haiku",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "anthropic/claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4.5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4",
        "knowledge": "2025-01-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 31999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 3,
          "cache_write": 37.5,
          "input": 30,
          "output": 150
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.7-fast",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 (Fast)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8-fast",
        "last_updated": "2026-05-27",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 (Fast)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-01-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "arcee-ai/coder-large": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 0.8
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "arcee-ai/coder-large",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-05",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Coder Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "arcee-ai/trinity-large-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.25,
          "output": 0.8
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "trinity",
        "id": "arcee-ai/trinity-large-thinking",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 262144,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Large Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/trinity-mini": {
        "attachment": false,
        "cost": {
          "input": 0.045,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "trinity-mini",
        "id": "arcee-ai/trinity-mini",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Mini",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/virtuoso-large": {
        "attachment": false,
        "cost": {
          "input": 0.75,
          "output": 1.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "arcee-ai/virtuoso-large",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-05",
        "limit": {
          "context": 131072,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Virtuoso Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "baidu/ernie-4.5-vl-424b-a47b": {
        "attachment": true,
        "cost": {
          "input": 0.42,
          "output": 1.25
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "ernie",
        "id": "baidu/ernie-4.5-vl-424b-a47b",
        "knowledge": "2025-03-31",
        "last_updated": "2025-06-30",
        "limit": {
          "context": 123000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 4.5 VL 424B A47B ",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "bytedance-seed/seed-1.6": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "bytedance-seed/seed-1.6",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "bytedance-seed/seed-1.6-flash": {
        "attachment": true,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "bytedance-seed/seed-1.6-flash",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "bytedance-seed/seed-2.0-lite": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "bytedance-seed/seed-2.0-lite",
        "last_updated": "2026-03-10",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed-2.0-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "bytedance-seed/seed-2.0-mini": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "bytedance-seed/seed-2.0-mini",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed-2.0-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "bytedance/ui-tars-1.5-7b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.1,
          "output": 0.2
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "bytedance/ui-tars-1.5-7b",
        "knowledge": "2025-01-31",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 128000,
          "output": 2048
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "UI-TARS 7B ",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "cognitivecomputations/dolphin-mistral-24b-venice-edition:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "cognitivecomputations/dolphin-mistral-24b-venice-edition:free",
        "knowledge": "2024-04-30",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Uncensored (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "cohere/command-a": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command-a",
        "id": "cohere/command-a",
        "knowledge": "2024-08-31",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "cohere/command-r-08-2024": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/command-r-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r-plus-08-2024": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere's RAG workhorse for long-context enterprise search and tool use",
        "family": "command-r",
        "id": "cohere/command-r-plus-08-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-08-30",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R+",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "cohere/command-r7b-12-2024": {
        "attachment": false,
        "cost": {
          "input": 0.0375,
          "output": 0.15
        },
        "description": "Cohere retrieval model for long-context chat and enterprise RAG workflows",
        "family": "command-r",
        "id": "cohere/command-r7b-12-2024",
        "knowledge": "2024-06-01",
        "last_updated": "2024-12-02",
        "limit": {
          "context": 128000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command R7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "cohere/north-mini-code:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Cohere coding model for practical software engineering and agentic edits",
        "family": "north",
        "id": "cohere/north-mini-code:free",
        "last_updated": "2026-06-17",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "North Mini Code (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-17",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepcogito/cogito-v2.1-671b": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 1.25
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "cogito",
        "id": "deepcogito/cogito-v2.1-671b",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cogito v2.1 671B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-chat": {
        "attachment": false,
        "cost": {
          "input": 0.2002,
          "output": 0.8001
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-chat",
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Chat",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat-v3-0324": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.24,
          "output": 0.9
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-chat-v3-0324",
        "knowledge": "2024-07-31",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 163840,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat-v3.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 0.21,
          "output": 0.79
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-chat-v3.1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.5
        },
        "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 64000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-0528": {
        "attachment": false,
        "cost": {
          "cache_read": 0.35,
          "input": 0.5,
          "output": 2.15
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek",
        "id": "deepseek/deepseek-r1-0528",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "R1 0528",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-r1-distill-llama-70b": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 0.8
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1-distill-llama-70b",
        "knowledge": "2024-07-31",
        "last_updated": "2025-01-23",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "R1 Distill Llama 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v3.1-terminus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "input": 0.27,
          "output": 0.95
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1-terminus",
        "knowledge": "2025-03-31",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 163840,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Terminus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02288,
          "input": 0.2288,
          "output": 0.3432
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-exp": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.41
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2-exp",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163840,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Exp",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.018,
          "input": 0.09,
          "output": 0.18
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1048576,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.083333,
          "input": 0.3,
          "output": 2.5,
          "reasoning": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-image": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.083333,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Nano Banana image model for fast generation, edits, and character-consistent assets",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash-image",
        "knowledge": "2025-06",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.083333,
          "input": 0.1,
          "output": 0.4,
          "reasoning": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite-preview-09-2025": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.083333,
          "input": 0.1,
          "output": 0.4,
          "reasoning": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite-preview-09-2025",
        "knowledge": "2025-01-31",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite Preview 09-2025",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 0.375,
          "input": 1.25,
          "output": 10,
          "reasoning": 10
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 0.375,
          "input": 1.25,
          "output": 10,
          "reasoning": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini",
        "id": "google/gemini-2.5-pro-preview",
        "knowledge": "2025-01-31",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro Preview 06-05",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro-preview-05-06": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "cache_write": 0.375,
          "input": 1.25,
          "output": 10,
          "reasoning": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro-preview-05-06",
        "knowledge": "2025-01-31",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro Preview 05-06",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.083333,
          "input": 0.5,
          "output": 3,
          "reasoning": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-image": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.375,
          "input": 2,
          "output": 12,
          "reasoning": 12
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini",
        "id": "google/gemini-3-pro-image",
        "last_updated": "2026-06-18",
        "limit": {
          "context": 65536,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Nano Banana Pro (Gemini 3 Pro Image)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-image-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.375,
          "input": 2,
          "output": 12,
          "reasoning": 12
        },
        "description": "Nano Banana Pro for higher-fidelity image generation and design-heavy edits",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 65536,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-image": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-image",
        "last_updated": "2026-06-18",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Nano Banana 2 (Gemini 3.1 Flash Image)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "minimal"
            ]
          }
        ],
        "release_date": "2026-06-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-image-preview": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini-flash",
        "id": "google/gemini-3.1-flash-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.083333,
          "input": 0.25,
          "output": 1.5,
          "reasoning": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite-image": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.5
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-lite-image",
        "knowledge": "2025-01-01",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 65536,
          "output": 66000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 0.083333,
          "input": 0.25,
          "output": 1.5,
          "reasoning": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.375,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "reasoning": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.375,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "reasoning": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview-customtools",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "cache_write": 0.083333,
          "input": 1.5,
          "output": 9,
          "reasoning": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-2-27b-it": {
        "attachment": false,
        "cost": {
          "input": 0.65,
          "output": 0.65
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-2-27b-it",
        "knowledge": "2024-06-30",
        "last_updated": "2024-07-13",
        "limit": {
          "context": 8192,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 2 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-3-12b-it": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.15
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3-12b-it",
        "knowledge": "2024-08-31",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.08,
          "output": 0.16
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3-27b-it",
        "knowledge": "2024-08-31",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3-4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.05,
          "output": 0.1
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3-4b-it",
        "knowledge": "2024-08-31",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-3n-e4b-it": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.12
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3n-e4b-it",
        "knowledge": "2024-08-31",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3n 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.06,
          "output": 0.33
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26b-a4b-it:free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26b-a4b-it:free",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B  (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09,
          "input": 0.12,
          "output": 0.35
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it:free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it:free",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "google/lyria-3-clip-preview": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "lyria",
        "id": "google/lyria-3-clip-preview",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Lyria 3 Clip Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "google/lyria-3-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "lyria",
        "id": "google/lyria-3-pro-preview",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Lyria 3 Pro Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gryphe/mythomax-l2-13b": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.06
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "gryphe/mythomax-l2-13b",
        "knowledge": "2023-06-30",
        "last_updated": "2023-07-02",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MythoMax 13B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-07-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "ibm-granite/granite-4.0-h-micro": {
        "attachment": false,
        "cost": {
          "input": 0.017,
          "output": 0.112
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "granite",
        "id": "ibm-granite/granite-4.0-h-micro",
        "last_updated": "2025-10-20",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Granite 4.0 Micro",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-20",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "ibm-granite/granite-4.1-8b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.05,
          "output": 0.1
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "granite",
        "id": "ibm-granite/granite-4.1-8b",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Granite 4.1 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inception/mercury-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "mercury",
        "id": "inception/mercury-2",
        "last_updated": "2026-03-04",
        "limit": {
          "context": 128000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.075,
          "output": 0.625
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "ling",
        "id": "inclusionai/ling-2.6-1t",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling-2.6-1T",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ling-2.6-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.002,
          "input": 0.01,
          "output": 0.03
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "ling",
        "id": "inclusionai/ling-2.6-flash",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling-2.6-flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ring-2.6-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.075,
          "output": 0.625
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "ring",
        "id": "inclusionai/ring-2.6-1t",
        "last_updated": "2026-05-08",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ring-2.6-1T",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "inflection/inflection-3-pi": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "inflection/inflection-3-pi",
        "knowledge": "2024-10-31",
        "last_updated": "2024-10-11",
        "limit": {
          "context": 8000,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inflection 3 Pi",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "inflection/inflection-3-productivity": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "inflection/inflection-3-productivity",
        "knowledge": "2024-10-31",
        "last_updated": "2024-10-11",
        "limit": {
          "context": 8000,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Inflection 3 Productivity",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "kwaipilot/kat-coder-pro-v2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "kat-coder",
        "id": "kwaipilot/kat-coder-pro-v2",
        "last_updated": "2026-03-27",
        "limit": {
          "context": 256000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KAT-Coder-Pro V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "liquid/lfm-2-24b-a2b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.12
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "liquid",
        "id": "liquid/lfm-2-24b-a2b",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LFM2-24B-A2B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "liquid/lfm-2.5-1.2b-instruct:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "family": "liquid",
        "id": "liquid/lfm-2.5-1.2b-instruct:free",
        "knowledge": "2025-06",
        "last_updated": "2026-01-20",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LFM2.5-1.2B-Instruct (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "liquid/lfm-2.5-1.2b-thinking:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "family": "liquid",
        "id": "liquid/lfm-2.5-1.2b-thinking:free",
        "knowledge": "2025-06",
        "last_updated": "2026-01-20",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LFM2.5-1.2B-Thinking (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mancer/weaver": {
        "attachment": false,
        "cost": {
          "input": 0.75,
          "output": 1
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "alpha",
        "id": "mancer/weaver",
        "knowledge": "2023-06-30",
        "last_updated": "2023-08-02",
        "limit": {
          "context": 8000,
          "output": 2000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Weaver (alpha)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-08-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.14
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3-8b-instruct",
        "knowledge": "2023-12-31",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.1-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.1-70b-instruct",
        "knowledge": "2023-12-31",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.03
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.1-8b-instruct",
        "knowledge": "2023-12-31",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3.2-11b-vision-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.345,
          "output": 0.345
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta-llama/llama-3.2-11b-vision-instruct",
        "knowledge": "2023-12-31",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11B Vision Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.2-1b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.027,
          "output": 0.201
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.2-1b-instruct",
        "knowledge": "2023-12-31",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 60000,
          "output": 60000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 1B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.2-3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.0509,
          "output": 0.335
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.2-3b-instruct",
        "knowledge": "2023-12-31",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 80000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.2-3b-instruct:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/llama-3.2-3b-instruct:free",
        "knowledge": "2023-12-31",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "meta-llama/llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.32
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "meta-llama/llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-3.3-70b-instruct:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "meta-llama/llama-3.3-70b-instruct:free",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 65536,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-4-maverick": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta-llama/llama-4-maverick",
        "knowledge": "2024-08-31",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 1048576,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Maverick",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-4-scout": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta-llama/llama-4-scout",
        "knowledge": "2024-08-31",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 327680,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/llama-guard-4-12b": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.18
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "meta-llama/llama-guard-4-12b",
        "knowledge": "2024-08-31",
        "last_updated": "2025-04-30",
        "limit": {
          "context": 163840,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama Guard 4 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "microsoft/phi-4": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.14
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "phi",
        "id": "microsoft/phi-4",
        "knowledge": "2024-06-30",
        "last_updated": "2025-01-10",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi 4",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "microsoft/wizardlm-2-8x22b": {
        "attachment": false,
        "cost": {
          "input": 0.62,
          "output": 0.62
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "microsoft/wizardlm-2-8x22b",
        "knowledge": "2024-04-30",
        "last_updated": "2024-04-16",
        "limit": {
          "context": 65535,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "WizardLM-2 8x22B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-16",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-01": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.1
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "minimax/minimax-01",
        "knowledge": "2024-03-31",
        "last_updated": "2025-01-15",
        "limit": {
          "context": 1000192,
          "output": 1000192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-01",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-m1": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m1",
        "knowledge": "2024-06-30",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1000000,
          "output": 40000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-17",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "cost": {
          "input": 0.255,
          "output": 1.02
        },
        "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2",
        "interleaved": {
          "field": "reasoning_details"
        },
        "last_updated": "2025-10-27",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2-her": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2-her",
        "last_updated": "2026-01-23",
        "limit": {
          "context": 65536,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2-her",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Earlier MiniMax agent model for practical coding and productivity tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.1",
        "interleaved": {
          "field": "reasoning_details"
        },
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.48
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax/minimax-m2.5",
        "interleaved": {
          "field": "reasoning_details"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "input": 0.18,
          "output": 0.72
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 524288,
          "output": 512000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/codestral-2508": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral coding model for code completion, generation, and developer workflows",
        "family": "codestral",
        "id": "mistralai/codestral-2508",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral 2508",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/devstral-2512": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes",
        "family": "devstral",
        "id": "mistralai/devstral-2512",
        "knowledge": "2025-12",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-09",
        "status": "deprecated",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/ministral-14b-2512": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-14b-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 14B 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/ministral-3b-2512": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-3b-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 3B 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/ministral-8b-2512": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.15
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistralai/ministral-8b-2512",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3 8B 2512",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistralai/mistral-large",
        "knowledge": "2024-11-30",
        "last_updated": "2024-02-26",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-2407": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 6
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistralai/mistral-large-2407",
        "knowledge": "2024-03-31",
        "last_updated": "2024-11-19",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 2407",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-large-2512": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "mistralai/mistral-large-2512",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/mistral-medium-3",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3-5": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/mistral-medium-3-5",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-medium-3.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistralai/mistral-medium-3.1",
        "knowledge": "2025-06-30",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 131072,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.03
        },
        "description": "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment",
        "family": "mistral-nemo",
        "id": "mistralai/mistral-nemo",
        "knowledge": "2024-07",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-saba": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 0.6
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "mistralai/mistral-saba",
        "knowledge": "2024-09-30",
        "last_updated": "2025-02-17",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Saba",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-small-24b-instruct-2501": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.08
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistralai/mistral-small-24b-instruct-2501",
        "knowledge": "2023-10-31",
        "last_updated": "2025-01-30",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "mistralai/mistral-small-2603": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents",
        "family": "mistral-small",
        "id": "mistralai/mistral-small-2603",
        "knowledge": "2025-06",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 4",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mistral-small-3.1-24b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.351,
          "output": 0.555
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistralai/mistral-small-3.1-24b-instruct",
        "knowledge": "2023-10-31",
        "last_updated": "2025-03-17",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.1 24B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-17",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "mistralai/mistral-small-3.2-24b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.075,
          "output": 0.2
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistralai/mistral-small-3.2-24b-instruct",
        "knowledge": "2023-10-31",
        "last_updated": "2025-06-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2 24B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/mixtral-8x22b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 6
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "mistralai/mixtral-8x22b-instruct",
        "knowledge": "2024-01-31",
        "last_updated": "2024-04-17",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mixtral 8x22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistralai/voxtral-small-24b-2507": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral",
        "id": "mistralai/voxtral-small-24b-2507",
        "last_updated": "2025-10-30",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voxtral Small 24B 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2": {
        "attachment": false,
        "cost": {
          "input": 0.57,
          "output": 2.3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2",
        "knowledge": "2024-12-31",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 131072,
          "output": 100352
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0711",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2-0905",
        "knowledge": "2024-12-31",
        "last_updated": "2025-09-04",
        "limit": {
          "context": 262144,
          "output": 100352
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262144,
          "output": 100352
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "input": 0.375,
          "output": 2.025
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.14,
          "input": 0.66,
          "output": 3.41
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 0.74,
          "output": 3.5
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "morph/morph-v3-fast": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "morph",
        "id": "morph/morph-v3-fast",
        "last_updated": "2025-07-07",
        "limit": {
          "context": 81920,
          "output": 38000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph V3 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "morph/morph-v3-large": {
        "attachment": false,
        "cost": {
          "input": 0.9,
          "output": 1.9
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "morph",
        "id": "morph/morph-v3-large",
        "last_updated": "2025-07-07",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph V3 Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "nex-agi/nex-n2-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "agi",
        "id": "nex-agi/nex-n2-pro",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nex-N2-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-3-llama-3.1-405b": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "nousresearch",
        "id": "nousresearch/hermes-3-llama-3.1-405b",
        "knowledge": "2023-12-31",
        "last_updated": "2024-08-16",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 3 405B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-3-llama-3.1-405b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "hermes",
        "id": "nousresearch/hermes-3-llama-3.1-405b:free",
        "knowledge": "2023-12-31",
        "last_updated": "2024-08-16",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 3 405B Instruct (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-16",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-3-llama-3.1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 0.7
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "nousresearch",
        "id": "nousresearch/hermes-3-llama-3.1-70b",
        "knowledge": "2023-12-31",
        "last_updated": "2024-08-18",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 3 70B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-4-405b": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "hermes",
        "id": "nousresearch/hermes-4-405b",
        "knowledge": "2024-08-31",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 405B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-26",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nousresearch/hermes-4-70b": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "hermes",
        "id": "nousresearch/hermes-4-70b",
        "knowledge": "2024-08-31",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 4 70B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-26",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nvidia/llama-3.3-nemotron-super-49b-v1.5": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 Nemotron Super 49B v1.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-nano-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-30b-a3b",
        "last_updated": "2025-12-15",
        "limit": {
          "context": 262144,
          "output": 228000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Nano 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-nano-30b-a3b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-30b-a3b:free",
        "last_updated": "2025-12-15",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Nano 30B A3B (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free",
        "last_updated": "2026-04-28",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Nano Omni (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.085,
          "output": 0.4
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super 120B A12B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b:free",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-ultra-550b-a55b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.5,
          "output": 2.2
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-ultra-550b-a55b",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra 550B A55B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-ultra-550b-a55b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-ultra-550b-a55b:free",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3.5-content-safety:free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "nemotron",
        "id": "nvidia/nemotron-3.5-content-safety:free",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3.5 Content Safety (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "nvidia/nemotron-nano-12b-v2-vl:free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "family": "nemotron",
        "id": "nvidia/nemotron-nano-12b-v2-vl:free",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Nano 12B 2 VL (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-nano-9b-v2:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-nano-9b-v2:free",
        "last_updated": "2025-08-18",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Nano 9B V2 (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo",
        "knowledge": "2021-09-01",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-0613": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo-0613",
        "knowledge": "2021-09-30",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 4095,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo (older v0613)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-16k": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo-16k",
        "knowledge": "2021-09-30",
        "last_updated": "2023-08-28",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo 16k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-08-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo-instruct",
        "knowledge": "2021-09-30",
        "last_updated": "2023-09-28",
        "limit": {
          "context": 4095,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4": {
        "attachment": false,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 8191,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-turbo-preview": {
        "attachment": false,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo-preview",
        "knowledge": "2023-12-31",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-05-13": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 15
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-05-13",
        "knowledge": "2023-09",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-05-13)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-08-06": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-08-06",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-08-06)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-11-20",
        "knowledge": "2023-09",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini-2024-07-18": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "o-mini",
        "id": "openai/gpt-4o-mini-2024-07-18",
        "knowledge": "2023-10-31",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o-mini (2024-07-18)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini-search-preview": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "o-mini",
        "id": "openai/gpt-4o-mini-search-preview",
        "knowledge": "2023-10-31",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o-mini Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-4o-search-preview": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-search-preview",
        "knowledge": "2023-10-31",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5-chat",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-5-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-image": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 10,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-5-image",
        "knowledge": "2024-10-01",
        "last_updated": "2025-10-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "GPT-5 Image",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-5-image-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 2
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-5-image-mini",
        "last_updated": "2025-10-16",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "GPT-5 Image Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "openai/gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-chat",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Codex GPT for repository edits, code review, and practical software agents",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-chat",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-10",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-10",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows",
        "family": "gpt-pro",
        "id": "openai/gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5.3-chat",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-image-2": {
        "attachment": true,
        "cost": {
          "cache_read": 2,
          "input": 8,
          "output": 15
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-5.4-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "image",
            "text"
          ]
        },
        "name": "GPT-5.4 Image 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt-mini",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt-pro",
        "id": "openai/gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "openai/gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-audio": {
        "attachment": true,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "openai/gpt-audio",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "GPT Audio",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-audio-mini": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "o-mini",
        "id": "openai/gpt-audio-mini",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "GPT Audio Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-chat-latest",
        "last_updated": "2026-05-05",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT Chat Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.15
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b:free",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.029,
          "output": 0.14
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "knowledge": "2024-06-30",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b:free",
        "knowledge": "2024-06-30",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-safeguard-20b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0375,
          "input": 0.075,
          "output": 0.3
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-safeguard-20b",
        "last_updated": "2025-10-29",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-safeguard-20b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o1-pro": {
        "attachment": true,
        "cost": {
          "input": 150,
          "output": 600
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "openai/o1-pro",
        "knowledge": "2023-09",
        "last_updated": "2025-03-19",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": false
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 2.5,
          "input": 10,
          "output": 40
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o",
        "id": "openai/o3-deep-research",
        "knowledge": "2024-05",
        "last_updated": "2024-06-26",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-06-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini-high": {
        "attachment": true,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o3-mini-high",
        "knowledge": "2023-10-31",
        "last_updated": "2025-02-12",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3 Mini High",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-02-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-pro": {
        "attachment": true,
        "cost": {
          "input": 20,
          "output": 80
        },
        "description": "High-effort o3 tier for difficult technical reasoning and careful answers",
        "family": "o-pro",
        "id": "openai/o3-pro",
        "knowledge": "2024-05",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "pdf",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o-mini",
        "id": "openai/o4-mini-deep-research",
        "knowledge": "2024-05",
        "last_updated": "2024-06-26",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-06-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/o4-mini-high": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o4-mini-high",
        "knowledge": "2024-06-30",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4 Mini High",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openrouter/auto": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "auto",
        "id": "openrouter/auto",
        "last_updated": "2023-11-08",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "pdf",
            "video"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Auto Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2023-11-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openrouter/bodybuilder": {
        "attachment": false,
        "description": "Preview model for early access evaluation, prototyping, and compatibility testing",
        "id": "openrouter/bodybuilder",
        "last_updated": "2025-12-05",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Body Builder (beta)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "openrouter/free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "openrouter/free",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 200000,
          "input": 200000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Free Models Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openrouter/fusion": {
        "attachment": false,
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "openrouter/fusion",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fusion",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-13",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "openrouter/pareto-code": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "openrouter/pareto-code",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 2000000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pareto Code Router",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "perceptron/perceptron-mk1": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 1.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "perceptron/perceptron-mk1",
        "last_updated": "2026-05-12",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perceptron Mk1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar": {
        "attachment": true,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar",
        "id": "perplexity/sonar",
        "last_updated": "2025-01-27",
        "limit": {
          "context": 127072,
          "output": 127072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-27",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-deep-research": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8,
          "reasoning": 3
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar-deep-research",
        "id": "perplexity/sonar-deep-research",
        "last_updated": "2025-03-07",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-pro": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "family": "sonar-pro",
        "id": "perplexity/sonar-pro",
        "last_updated": "2025-03-07",
        "limit": {
          "context": 200000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-pro-search": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "family": "sonar-pro",
        "id": "perplexity/sonar-pro-search",
        "last_updated": "2025-10-30",
        "limit": {
          "context": 200000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Pro Search",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar-reasoning-pro": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Web-grounded reasoning model for multi-step research and cited answers",
        "family": "sonar-reasoning",
        "id": "perplexity/sonar-reasoning-pro",
        "last_updated": "2025-03-07",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "poolside/laguna-m.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.2,
          "output": 0.4
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "poolside/laguna-m.1",
        "last_updated": "2026-04-28",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna M.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-m.1:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "id": "poolside/laguna-m.1:free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-28",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna M.1 (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-xs-2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.06,
          "output": 0.12
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "poolside/laguna-xs-2.1",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna XS 2.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-07-02",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-xs-2.1:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "id": "poolside/laguna-xs-2.1:free",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna XS 2.1 (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-07-02",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-xs.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.1,
          "output": 0.2
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "poolside/laguna-xs.2",
        "last_updated": "2026-04-28",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna XS.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-xs.2:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
        "id": "poolside/laguna-xs.2:free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-28",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna XS.2 (free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-72b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.36,
          "output": 0.4
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen-2.5-72b-instruct",
        "knowledge": "2024-06-30",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen-2.5-7b-instruct",
        "knowledge": "2024-06-30",
        "last_updated": "2024-10-16",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 7B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-2.5-coder-32b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.66,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen-2.5-coder-32b-instruct",
        "knowledge": "2024-06-30",
        "last_updated": "2024-11-11",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 Coder 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.052,
          "cache_write": 0.325,
          "input": 0.26,
          "output": 0.78
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen-plus",
        "knowledge": "2024-04",
        "last_updated": "2025-09-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-plus-2025-07-28": {
        "attachment": false,
        "cost": {
          "input": 0.26,
          "output": 0.78
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen-plus-2025-07-28",
        "knowledge": "2025-03-31",
        "last_updated": "2025-09-08",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus 0728",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen-plus-2025-07-28:thinking": {
        "attachment": false,
        "cost": {
          "cache_write": 0.325,
          "input": 0.26,
          "output": 0.78
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen-plus-2025-07-28:thinking",
        "knowledge": "2025-03-31",
        "last_updated": "2025-09-08",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen Plus 0728 (thinking)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen2.5-vl-72b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4,
          "input": 0.8,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen2.5-vl-72b-instruct",
        "knowledge": "2024-06-30",
        "last_updated": "2025-02-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5 VL 72B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-02-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "qwen/qwen3-14b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.24
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-14b",
        "knowledge": "2025-03-31",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 40960,
          "output": 40960
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 14B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b": {
        "attachment": false,
        "cost": {
          "input": 0.455,
          "output": 1.82
        },
        "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B-A22B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-2507": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-2507",
        "knowledge": "2025-06-30",
        "last_updated": "2025-07-21",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1495,
          "output": 1.495
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-235b-a22b-thinking-2507",
        "knowledge": "2025-06-30",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 131072,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-30b-a3b",
        "knowledge": "2025-03-31",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 40960,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.04815,
          "output": 0.19305
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-30b-a3b-instruct-2507",
        "knowledge": "2025-06-30",
        "last_updated": "2025-07-29",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-30b-a3b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 1.56
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-30b-a3b-thinking-2507",
        "knowledge": "2025-06-30",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 81920,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.28
        },
        "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-32b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 40960,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-8b": {
        "attachment": false,
        "cost": {
          "input": 0.117,
          "output": 0.455
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-8b",
        "knowledge": "2025-03-31",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 8B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 1.8
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder",
        "knowledge": "2025-06-30",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.27
        },
        "description": "Smaller Qwen coder for efficient local agents and repo-level fixes",
        "family": "qwen",
        "id": "qwen/qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 160000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.039,
          "cache_write": 0.24375,
          "input": 0.195,
          "output": 0.975
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-flash",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "cache_read": 0.07,
          "input": 0.11,
          "output": 0.8
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder-next",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.13,
          "cache_write": 0.8125,
          "input": 0.65,
          "output": 3.25
        },
        "description": "Hosted Qwen coder for software agents, repo edits, and long-context code",
        "family": "qwen",
        "id": "qwen/qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-coder:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen/qwen3-coder:free",
        "knowledge": "2025-06-30",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.156,
          "cache_write": 0.975,
          "input": 0.78,
          "output": 3.9
        },
        "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.78,
          "output": 3.9
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen/qwen3-max-thinking",
        "last_updated": "2026-02-09",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 1.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-next-80b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-instruct:free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-next-80b-a3b-instruct:free",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Instruct (free)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.0975,
          "output": 0.78
        },
        "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents",
        "family": "qwen",
        "id": "qwen/qwen3-next-80b-a3b-thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Next 80B-A3B (Thinking)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-235b-a22b-instruct": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.2,
          "output": 0.88
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-235b-a22b-instruct",
        "knowledge": "2025-03-31",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-235b-a22b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.26,
          "output": 2.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-235b-a22b-thinking",
        "knowledge": "2025-03-31",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-30b-a3b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.52
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-30b-a3b-instruct",
        "knowledge": "2025-03-31",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 30B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-30b-a3b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 1.56
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-30b-a3b-thinking",
        "knowledge": "2025-03-31",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 30B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-32b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.104,
          "output": 0.416
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-32b-instruct",
        "last_updated": "2025-10-23",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 32B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-8b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.117,
          "output": 0.455
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-8b-instruct",
        "last_updated": "2025-10-14",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 8B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-vl-8b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.117,
          "output": 1.365
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3-vl-8b-thinking",
        "last_updated": "2025-10-14",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 8B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0.26,
          "output": 2.08
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-122b-a10b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.195,
          "output": 1.56
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-27b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-35b-a3b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.14,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-35b-a3b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 81920
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.385,
          "output": 2.45
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-397b-a17b",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 131072,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-9b": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.15
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3.5-9b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-flash-02-23": {
        "attachment": true,
        "cost": {
          "input": 0.065,
          "output": 0.26
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-flash-02-23",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-plus-02-15": {
        "attachment": true,
        "cost": {
          "input": 0.26,
          "output": 1.56
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-plus-02-15",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus 2026-02-15",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-plus-20260420": {
        "attachment": true,
        "cost": {
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.5",
        "id": "qwen/qwen3.5-plus-20260420",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus 2026-04-20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.285,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.6-27b",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262140,
          "output": 262140
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 1
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "qwen/qwen3.6-35b-a3b",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-flash": {
        "attachment": true,
        "cost": {
          "cache_write": 0.234375,
          "input": 0.1875,
          "output": 1.125
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "qwen/qwen3.6-flash",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-max-preview": {
        "attachment": false,
        "cost": {
          "cache_write": 1.3,
          "input": 1.04,
          "output": 6.24
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "qwen/qwen3.6-max-preview",
        "knowledge": "2025-04",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Max Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-plus": {
        "attachment": true,
        "cost": {
          "cache_write": 0.40625,
          "input": 0.325,
          "output": 1.95
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen/qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 1.5625,
          "input": 1.25,
          "output": 3.75
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen/qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.7-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.064,
          "cache_write": 0.4,
          "input": 0.32,
          "output": 1.28
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen/qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "rekaai/reka-edge": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "reka",
        "id": "rekaai/reka-edge",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Reka Edge",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "rekaai/reka-flash-3": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "reka",
        "id": "rekaai/reka-flash-3",
        "knowledge": "2025-01-31",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Reka Flash 3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "relace/relace-apply-3": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 1.25
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "id": "relace/relace-apply-3",
        "last_updated": "2025-09-26",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Relace Apply 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-26",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "relace/relace-search": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "relace/relace-search",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Relace Search",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "sakana/fugu-ultra": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Quality-first multi-agent model for hard research, analysis, and competitions",
        "family": "fugu",
        "id": "sakana/fugu-ultra",
        "last_updated": "2026-06-24",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fugu Ultra",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "max",
              "xhigh",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "sao10k/l3-lunaris-8b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.05
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "sao10k/l3-lunaris-8b",
        "knowledge": "2023-12-31",
        "last_updated": "2024-08-13",
        "limit": {
          "context": 8192,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3 8B Lunaris",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "sao10k/l3.1-70b-hanami-x1": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 3
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "sao10k/l3.1-70b-hanami-x1",
        "knowledge": "2023-12-31",
        "last_updated": "2025-01-08",
        "limit": {
          "context": 16000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Hanami x1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "sao10k/l3.1-euryale-70b": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 0.85
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "sao10k/l3.1-euryale-70b",
        "knowledge": "2023-12-31",
        "last_updated": "2024-08-28",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 Euryale 70B v2.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-08-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "sao10k/l3.3-euryale-70b": {
        "attachment": false,
        "cost": {
          "input": 0.65,
          "output": 0.75
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "sao10k/l3.3-euryale-70b",
        "knowledge": "2023-12-31",
        "last_updated": "2024-12-18",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 Euryale 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "stepfun/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "id": "stepfun/step-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.7-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.2,
          "output": 1.15
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "id": "stepfun/step-3.7-flash",
        "knowledge": "2026-01-01",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "switchpoint/router": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 3.4
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "id": "switchpoint/router",
        "last_updated": "2025-07-11",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Switchpoint Router",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "tencent/hunyuan-a13b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "tencent/hunyuan-a13b-instruct",
        "knowledge": "2025-03-31",
        "last_updated": "2025-07-08",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hunyuan A13B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "tencent/hy3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.021,
          "input": 0.063,
          "output": 0.21
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "Hy",
        "id": "tencent/hy3-preview",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3 preview",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-20",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "thedrummer/cydonia-24b-v4.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.3,
          "output": 0.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/cydonia-24b-v4.1",
        "knowledge": "2024-04-30",
        "last_updated": "2025-09-27",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cydonia 24B V4.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "thedrummer/rocinante-12b": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.5
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/rocinante-12b",
        "knowledge": "2024-04-30",
        "last_updated": "2024-09-30",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Rocinante 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "thedrummer/skyfall-36b-v2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "input": 0.55,
          "output": 0.8
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/skyfall-36b-v2",
        "knowledge": "2024-06-30",
        "last_updated": "2025-03-10",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Skyfall 36B V2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "thedrummer/unslopnemo-12b": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 0.4
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "thedrummer/unslopnemo-12b",
        "knowledge": "2024-04-30",
        "last_updated": "2024-11-08",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "UnslopNemo 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "undi95/remm-slerp-l2-13b": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 0.65
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "undi95/remm-slerp-l2-13b",
        "knowledge": "2023-06-30",
        "last_updated": "2023-07-22",
        "limit": {
          "context": 6144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ReMM SLERP 13B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-07-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "upstage/solar-pro-3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "solar-pro",
        "id": "upstage/solar-pro-3",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Solar Pro 3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "writer/palmyra-x5": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 6
        },
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "palmyra",
        "id": "writer/palmyra-x5",
        "last_updated": "2026-01-21",
        "limit": {
          "context": 1040000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Palmyra X5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-21",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "x-ai/grok-4.20": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "x-ai/grok-4.20",
        "knowledge": "2025-09-01",
        "last_updated": "2026-03-31",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.20-multi-agent": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "x-ai/grok-4.20-multi-agent",
        "knowledge": "2025-09-01",
        "last_updated": "2026-03-31",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Multi-Agent",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "x-ai/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "x-ai/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 2
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "x-ai/grok-build-0.1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5": {
        "attachment": true,
        "cost": {
          "input": 0.105,
          "output": 0.28
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "xiaomi/mimo-v2.5",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 32000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0036,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "xiaomi/mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "z-ai/glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.025,
          "input": 0.13,
          "output": 0.85
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "z-ai/glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "z-ai/glm-4.5v",
        "knowledge": "2025-04",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 65536,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.43,
          "output": 1.74
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "z-ai/glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.055,
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "z-ai/glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 1.75
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "z-ai/glm-4.7",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.06,
          "output": 0.4
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm-flash",
        "id": "z-ai/glm-4.7-flash",
        "interleaved": {
          "field": "reasoning_details"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 1.92
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "z-ai/glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202752,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Faster GLM-5 lane for coding agents that need lower latency",
        "family": "glm",
        "id": "z-ai/glm-5-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-16",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-16",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1794,
          "input": 0.966,
          "output": 3.036
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "z-ai/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.18,
          "input": 0.93,
          "output": 3
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "z-ai/glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1048576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "z-ai/glm-5v-turbo",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "~anthropic/claude-fable-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "~anthropic/claude-fable-latest",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "~anthropic/claude-haiku-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "~anthropic/claude-haiku-latest",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic Claude Haiku Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 63999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "~anthropic/claude-opus-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "~anthropic/claude-opus-latest",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "~anthropic/claude-sonnet-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "~anthropic/claude-sonnet-latest",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Anthropic Claude Sonnet Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "max": 127999,
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "~google/gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "cache_write": 0.083333,
          "input": 1.5,
          "output": 9,
          "reasoning": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "~google/gemini-flash-latest",
        "knowledge": "2025-01-01",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "~google/gemini-pro-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0.375,
          "input": 2,
          "output": 12,
          "reasoning": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "~google/gemini-pro-latest",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "audio",
            "pdf",
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemini Pro Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "~moonshotai/kimi-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.14,
          "input": 0.66,
          "output": 3.41
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi",
        "id": "~moonshotai/kimi-latest",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MoonshotAI Kimi Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "~openai/gpt-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "~openai/gpt-latest",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "~openai/gpt-mini-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "~openai/gpt-mini-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2026-04-27",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT Mini Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "OpenRouter",
    "npm": "@openrouter/ai-sdk-provider"
  },
  "orcarouter": {
    "api": "https://api.orcarouter.ai/v1",
    "doc": "https://docs.orcarouter.ai",
    "env": [
      "ORCAROUTER_API_KEY"
    ],
    "id": "orcarouter",
    "models": {
      "anthropic/claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4.5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-chat",
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Chat",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-reasoner": {
        "attachment": true,
        "cost": {
          "cache_read": 0.028,
          "input": 0.435,
          "output": 0.87
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-reasoner",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-09",
        "last_updated": "2026-02-28",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek Reasoner",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.19,
          "output": 0.37
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.56,
          "output": 1.12
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "input_audio": 0.3,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 4,
          "output": 18,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "input_audio": 0.5,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 4,
          "output": 18,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview-customtools": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 4,
          "output": 18,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview-customtools",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview Custom Tools",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-flash-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.5,
          "input_audio": 1,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-flash-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-flash-lite-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-flash-lite-latest",
        "knowledge": "2025-01",
        "last_updated": "2025-09-25",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Flash-Lite Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.06,
          "output": 0.33
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.13,
          "output": 0.38
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "grok/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi/kimi-k2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax/minimax-m2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7-highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo",
        "knowledge": "2021-09-01",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 60
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4",
        "knowledge": "2023-11",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-05-13": {
        "attachment": true,
        "cost": {
          "input": 5,
          "output": 15
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-05-13",
        "knowledge": "2023-09",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-05-13)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-08-06": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-08-06",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-08-06)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-2024-11-20",
        "knowledge": "2023-09",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o (2024-11-20)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-5-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "openai/gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-chat-latest",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Codex GPT for repository edits, code review, and practical software agents",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows",
        "family": "gpt-pro",
        "id": "openai/gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-chat-latest": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5.3-chat-latest",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat (latest)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 5,
          "output": 22.5,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.15,
                "input": 1.5,
                "output": 9
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt-mini",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt-nano",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 60,
          "output": 270,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt-pro",
        "id": "openai/gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "openai/gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "orcarouter/auto": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "auto",
        "id": "orcarouter/auto",
        "last_updated": "2026-05-14",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OrcaRouter Auto",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 0.359,
          "output": 1.434
        },
        "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
        "family": "qwen",
        "id": "qwen/qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-122b-a10b": {
        "attachment": true,
        "cost": {
          "input": 0.115,
          "output": 0.917
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-122b-a10b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-27b": {
        "attachment": true,
        "cost": {
          "input": 0.086,
          "output": 0.688
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-27b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.057,
          "output": 0.459
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-35b-a3b",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.172,
          "output": 1.032
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-397b-a17b",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-plus": {
        "attachment": false,
        "cost": {
          "input": 0.115,
          "output": 0.688,
          "reasoning": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen/qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.248,
          "output": 1.485
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "qwen/qwen3.6-35b-a3b",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen/qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "z-ai/glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0,
          "input": 0.2,
          "output": 1.1
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "z-ai/glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "z-ai/glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "z-ai/glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 1,
          "output": 3.2
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "z-ai/glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "z-ai/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "OrcaRouter",
    "npm": "@ai-sdk/openai-compatible"
  },
  "ovhcloud": {
    "api": "https://oai.endpoints.kepler.ai.cloud.ovh.net/v1",
    "doc": "https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog//",
    "env": [
      "OVHCLOUD_API_KEY"
    ],
    "id": "ovhcloud",
    "models": {
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.47
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-28",
        "structured_output": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.18
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "gpt-oss-20b",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-28",
        "structured_output": true,
        "tool_call": true
      },
      "meta-llama-3_3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.74,
          "output": 0.74
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "meta-llama-3_3-70b-instruct",
        "last_updated": "2025-04-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3_3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-7b-instruct-v0.3": {
        "attachment": false,
        "cost": {
          "input": 0.11,
          "output": 0.11
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistral-7b-instruct-v0.3",
        "last_updated": "2025-04-01",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral-7B-Instruct-v0.3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-nemo-instruct-2407": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.14
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "id": "mistral-nemo-instruct-2407",
        "last_updated": "2024-11-20",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral-Nemo-Instruct-2407",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-3.2-24b-instruct-2506": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.31
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "id": "mistral-small-3.2-24b-instruct-2506",
        "last_updated": "2025-07-16",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral-Small-3.2-24B-Instruct-2506",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen2.5-vl-72b-instruct": {
        "attachment": true,
        "cost": {
          "input": 1.01,
          "output": 1.01
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "qwen2.5-vl-72b-instruct",
        "last_updated": "2025-03-31",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-VL-72B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "qwen3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.25
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-32b",
        "last_updated": "2025-07-16",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-32B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.26
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "qwen3-coder-30b-a3b-instruct",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-30B-A3B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.71,
          "output": 4.25
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "qwen3.5-397b-a17b",
        "last_updated": "2026-05-18",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-9b": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0.18
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "qwen3.5-9b",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.47,
          "output": 3.19
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "qwen3.6-27b",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6-27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3guard-gen-0.6b": {
        "attachment": false,
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "qwen3guard-gen-0.6b",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3Guard-Gen-0.6B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-22",
        "temperature": true,
        "tool_call": false
      },
      "qwen3guard-gen-8b": {
        "attachment": false,
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "id": "qwen3guard-gen-8b",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 32768,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3Guard-Gen-8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-22",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "OVHcloud AI Endpoints",
    "npm": "@ai-sdk/openai-compatible"
  },
  "perplexity": {
    "doc": "https://docs.perplexity.ai",
    "env": [
      "PERPLEXITY_API_KEY"
    ],
    "id": "perplexity",
    "models": {
      "sonar": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Fast web-grounded Sonar for current answers, citations, and lightweight retrieval",
        "family": "sonar",
        "id": "sonar",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "sonar-deep-research": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8,
          "reasoning": 3
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "id": "sonar-deep-research",
        "knowledge": "2025-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Perplexity Sonar Deep Research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-02-01",
        "temperature": false,
        "tool_call": false
      },
      "sonar-pro": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Deeper Sonar search model with broader retrieval and stronger synthesis",
        "family": "sonar-pro",
        "id": "sonar-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "sonar-reasoning-pro": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 8
        },
        "description": "Web-grounded Sonar for multi-step research questions that need cited reasoning",
        "family": "sonar-reasoning",
        "id": "sonar-reasoning-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Perplexity",
    "npm": "@ai-sdk/perplexity"
  },
  "perplexity-agent": {
    "api": "https://api.perplexity.ai/v1",
    "doc": "https://docs.perplexity.ai/docs/agent-api/models",
    "env": [
      "PERPLEXITY_API_KEY"
    ],
    "id": "perplexity-agent",
    "models": {
      "anthropic/claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "context_over_200k": {
            "cache_read": 0.05,
            "input": 0.5,
            "output": 3
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.05,
              "input": 0.5,
              "output": 3,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 2.5
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "knowledge": "2026-02",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 1000000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "temperature": false,
        "tool_call": true
      },
      "perplexity/sonar": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0625,
          "input": 0.25,
          "output": 2.5
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar",
        "id": "perplexity/sonar",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4-1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4-1-fast-non-reasoning",
        "knowledge": "2025-07",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-19",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Perplexity Agent",
    "npm": "@ai-sdk/openai"
  },
  "poe": {
    "api": "https://api.poe.com/v1",
    "doc": "https://creator.poe.com/docs/external-applications/openai-compatible-api",
    "env": [
      "POE_API_KEY"
    ],
    "id": "poe",
    "models": {
      "anthropic/claude-haiku-3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.021,
          "cache_write": 0.26,
          "input": 0.21,
          "output": 1.1
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-3",
        "last_updated": "2024-03-09",
        "limit": {
          "context": 189096,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Haiku-3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-09",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-haiku-3.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.068,
          "cache_write": 0.85,
          "input": 0.68,
          "output": 3.4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-3.5",
        "last_updated": "2024-10-01",
        "limit": {
          "context": 189096,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Haiku-3.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.085,
          "cache_write": 1.1,
          "input": 0.85,
          "output": 4.3
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4.5",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 192000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Haiku-4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 63999,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.3,
          "cache_write": 16,
          "input": 13,
          "output": 64
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4",
        "last_updated": "2025-05-21",
        "limit": {
          "context": 192512,
          "output": 28672
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Opus-4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-21",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.3,
          "cache_write": 16,
          "input": 13,
          "output": 64
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.1",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 196608,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Opus-4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 31999,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.43,
          "cache_write": 5.3,
          "input": 4.3,
          "output": 21
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.5",
        "last_updated": "2025-11-21",
        "limit": {
          "context": 196608,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Opus-4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "max": 63999,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-21",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.43,
          "cache_write": 5.3,
          "input": 4.3,
          "output": 21
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.6",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 983040,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Opus-4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-04",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.43,
          "cache_write": 5.4,
          "input": 4.3,
          "output": 21
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.7",
        "last_updated": "2026-04-15",
        "limit": {
          "context": 1048576,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Opus-4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-15",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "input": 4.2929,
          "output": 21.4646
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1048576,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Opus-4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-3.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 3.2,
          "input": 2.6,
          "output": 13
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-3.5",
        "last_updated": "2024-06-05",
        "limit": {
          "context": 189096,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Sonnet-3.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-05",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-3.5-june": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 3.2,
          "input": 2.6,
          "output": 13
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-3.5-june",
        "last_updated": "2024-11-18",
        "limit": {
          "context": 189096,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Sonnet-3.5-June",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-18",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-3.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 3.2,
          "input": 2.6,
          "output": 13
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-3.7",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 196608,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Sonnet-3.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-02-19",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 3.2,
          "input": 2.6,
          "output": 13
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "last_updated": "2025-05-21",
        "limit": {
          "context": 983040,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Sonnet-4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-21",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 3.2,
          "input": 2.6,
          "output": 13
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.5",
        "last_updated": "2025-09-26",
        "limit": {
          "context": 983040,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Sonnet-4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 31999,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-26",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 3.2,
          "input": 2.6,
          "output": 13
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4.6",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 983040,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude-Sonnet-4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "temperature": false,
        "tool_call": true
      },
      "cerebras/gpt-oss-120b-cs": {
        "attachment": true,
        "cost": {
          "input": 0.35,
          "output": 0.75
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "cerebras/gpt-oss-120b-cs",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS-120B-CS",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-06",
        "temperature": false,
        "tool_call": true
      },
      "cerebras/llama-3.1-8b-cs": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "id": "cerebras/llama-3.1-8b-cs",
        "last_updated": "2025-05-13",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.1-8B-CS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-13",
        "temperature": false,
        "tool_call": true
      },
      "cerebras/llama-3.3-70b-cs": {
        "attachment": true,
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "cerebras/llama-3.3-70b-cs",
        "last_updated": "2025-05-13",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "llama-3.3-70b-cs",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-13",
        "status": "deprecated",
        "temperature": false,
        "tool_call": false
      },
      "cerebras/qwen3-235b-2507-cs": {
        "attachment": true,
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "cerebras/qwen3-235b-2507-cs",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-235b-2507-cs",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-06",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "cerebras/qwen3-32b-cs": {
        "attachment": true,
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "cerebras/qwen3-32b-cs",
        "last_updated": "2025-05-15",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "qwen3-32b-cs",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-15",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "elevenlabs/elevenlabs-music": {
        "attachment": true,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "elevenlabs",
        "id": "elevenlabs/elevenlabs-music",
        "last_updated": "2025-08-29",
        "limit": {
          "context": 2000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "ElevenLabs-Music",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-29",
        "temperature": false,
        "tool_call": true
      },
      "elevenlabs/elevenlabs-v2.5-turbo": {
        "attachment": true,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "elevenlabs",
        "id": "elevenlabs/elevenlabs-v2.5-turbo",
        "last_updated": "2024-10-28",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "ElevenLabs-v2.5-Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-28",
        "temperature": false,
        "tool_call": true
      },
      "elevenlabs/elevenlabs-v3": {
        "attachment": true,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "elevenlabs",
        "id": "elevenlabs/elevenlabs-v3",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "ElevenLabs-v3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-05",
        "temperature": false,
        "tool_call": true
      },
      "empiriolabs/deepseek-v4-flash-el": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "id": "empiriolabs/deepseek-v4-flash-el",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V4-Flash-EL",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "tool_call": true
      },
      "empiriolabs/deepseek-v4-pro-el": {
        "attachment": true,
        "cost": {
          "input": 1.67,
          "output": 3.33
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "id": "empiriolabs/deepseek-v4-pro-el",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 1000000,
          "input": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V4-Pro-EL",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "tool_call": true
      },
      "fireworks-ai/kimi-k2.5-fw": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "fireworks-ai/kimi-k2.5-fw",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "input": 245760,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5-FW",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-27",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-2.0-flash": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.42
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-2.0-flash",
        "last_updated": "2024-12-11",
        "limit": {
          "context": 990000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-2.0-Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-11",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-2.0-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.052,
          "output": 0.21
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.0-flash-lite",
        "last_updated": "2025-02-05",
        "limit": {
          "context": 990000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-2.0-Flash-Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-05",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.021,
          "input": 0.21,
          "output": 1.8
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "last_updated": "2025-04-26",
        "limit": {
          "context": 1065535,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-2.5-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-26",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "last_updated": "2025-06-19",
        "limit": {
          "context": 1024000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-2.5-Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-19",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.087,
          "input": 0.87,
          "output": 7
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "last_updated": "2025-02-05",
        "limit": {
          "context": 1065535,
          "output": 65535
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-2.5-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-02-05",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-3-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.4,
          "output": 2.4
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3-flash",
        "last_updated": "2025-10-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-3-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-07",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-3-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 1.6,
          "output": 9.6
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro",
        "last_updated": "2025-10-22",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-3-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-22",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-3.1-flash-lite",
        "last_updated": "2026-02-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-3.1-Flash-Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-18",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-3.1-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-3.1-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1515,
          "input": 1.5152,
          "output": 9.0909
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini-3.5-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "google/gemini-deep-research": {
        "attachment": true,
        "cost": {
          "input": 1.6,
          "output": 9.6
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "google/gemini-deep-research",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 1048576,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "google/gemma-4-31b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "id": "google/gemma-4-31b",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma-4-31B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-02",
        "temperature": false,
        "tool_call": true
      },
      "google/imagen-3": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-3",
        "last_updated": "2024-10-15",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen-3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-15",
        "temperature": false,
        "tool_call": true
      },
      "google/imagen-3-fast": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-3-fast",
        "last_updated": "2024-10-17",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen-3-Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-17",
        "temperature": false,
        "tool_call": true
      },
      "google/imagen-4": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-22",
        "temperature": false,
        "tool_call": true
      },
      "google/imagen-4-fast": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4-fast",
        "last_updated": "2025-06-25",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen-4-Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-25",
        "temperature": false,
        "tool_call": true
      },
      "google/imagen-4-ultra": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4-ultra",
        "last_updated": "2025-05-24",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen-4-Ultra",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-24",
        "temperature": false,
        "tool_call": true
      },
      "google/lyria": {
        "attachment": true,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "lyria",
        "id": "google/lyria",
        "last_updated": "2025-06-04",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Lyria",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-04",
        "temperature": false,
        "tool_call": true
      },
      "google/nano-banana": {
        "attachment": true,
        "cost": {
          "cache_read": 0.021,
          "input": 0.21,
          "output": 1.8
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "nano-banana",
        "id": "google/nano-banana",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 65536,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano-Banana",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-21",
        "temperature": false,
        "tool_call": true
      },
      "google/nano-banana-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "nano-banana",
        "id": "google/nano-banana-pro",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 65536,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Nano-Banana-Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-19",
        "temperature": false,
        "tool_call": true
      },
      "google/veo-2": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-2",
        "last_updated": "2024-12-02",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-02",
        "temperature": false,
        "tool_call": true
      },
      "google/veo-3": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3",
        "last_updated": "2025-05-21",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo-3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-21",
        "temperature": false,
        "tool_call": true
      },
      "google/veo-3-fast": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3-fast",
        "last_updated": "2025-10-13",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo-3-Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-13",
        "temperature": false,
        "tool_call": true
      },
      "google/veo-3.1": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3.1",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo-3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": false,
        "tool_call": true
      },
      "google/veo-3.1-fast": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3.1-fast",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo-3.1-Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": false,
        "tool_call": true
      },
      "ideogramai/ideogram": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ideogram",
        "id": "ideogramai/ideogram",
        "last_updated": "2024-04-03",
        "limit": {
          "context": 150,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Ideogram",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-03",
        "temperature": false,
        "tool_call": true
      },
      "ideogramai/ideogram-v2": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ideogram",
        "id": "ideogramai/ideogram-v2",
        "last_updated": "2024-08-21",
        "limit": {
          "context": 150,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Ideogram-v2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-21",
        "temperature": false,
        "tool_call": true
      },
      "ideogramai/ideogram-v2a": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ideogram",
        "id": "ideogramai/ideogram-v2a",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 150,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Ideogram-v2a",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "temperature": false,
        "tool_call": true
      },
      "ideogramai/ideogram-v2a-turbo": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ideogram",
        "id": "ideogramai/ideogram-v2a-turbo",
        "last_updated": "2025-02-27",
        "limit": {
          "context": 150,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Ideogram-v2a-Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-27",
        "temperature": false,
        "tool_call": true
      },
      "lumalabs/ray2": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "ray",
        "id": "lumalabs/ray2",
        "last_updated": "2025-02-20",
        "limit": {
          "context": 5000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Ray2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-20",
        "temperature": false,
        "tool_call": true
      },
      "novita/deepseek-v3.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 0.27,
          "output": 0.4
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "novita/deepseek-v3.2",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "novita/glm-4.6": {
        "attachment": true,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "novita/glm-4.6",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-30",
        "temperature": false,
        "tool_call": true
      },
      "novita/glm-4.6v": {
        "attachment": true,
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "novita/glm-4.6v",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 131000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.6v",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-09",
        "temperature": false,
        "tool_call": true
      },
      "novita/glm-4.7": {
        "attachment": true,
        "description": "Legacy model retained for compatibility with older integrations",
        "id": "novita/glm-4.7",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 205000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "novita/glm-4.7-flash": {
        "attachment": true,
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "novita/glm-4.7-flash",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 65500
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.7-flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": false,
        "tool_call": true
      },
      "novita/glm-4.7-n": {
        "attachment": true,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "novita/glm-4.7-n",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 205000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "glm-4.7-n",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": false,
        "tool_call": true
      },
      "novita/glm-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "novita/glm-5",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 205000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-15",
        "temperature": true,
        "tool_call": true
      },
      "novita/kimi-k2-thinking": {
        "attachment": true,
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "family": "kimi-thinking",
        "id": "novita/kimi-k2-thinking",
        "last_updated": "2025-11-07",
        "limit": {
          "context": 256000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "kimi-k2-thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-11-07",
        "temperature": false,
        "tool_call": true
      },
      "novita/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "novita/kimi-k2.5",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 128000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "novita/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.96,
          "output": 4.04
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "novita/kimi-k2.6",
        "knowledge": "2025-04",
        "last_updated": "2026-05-02",
        "limit": {
          "context": 262144,
          "input": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "novita/minimax-m2.1": {
        "attachment": true,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "novita/minimax-m2.1",
        "last_updated": "2025-12-26",
        "limit": {
          "context": 205000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "minimax-m2.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-26",
        "temperature": false,
        "tool_call": true
      },
      "openai/chatgpt-4o-latest": {
        "attachment": true,
        "cost": {
          "input": 4.5,
          "output": 14
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gpt",
        "id": "openai/chatgpt-4o-latest",
        "last_updated": "2024-08-14",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ChatGPT-4o-Latest",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-14",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "openai/dall-e-3": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "dall-e",
        "id": "openai/dall-e-3",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 800,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "DALL-E-3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo": {
        "attachment": true,
        "cost": {
          "input": 0.45,
          "output": 1.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo",
        "last_updated": "2023-09-13",
        "limit": {
          "context": 16384,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-instruct": {
        "attachment": true,
        "cost": {
          "input": 1.4,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo-instruct",
        "last_updated": "2023-09-20",
        "limit": {
          "context": 3500,
          "output": 1024
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-Turbo-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-20",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo-raw": {
        "attachment": true,
        "cost": {
          "input": 0.45,
          "output": 1.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo-raw",
        "last_updated": "2023-09-27",
        "limit": {
          "context": 4524,
          "output": 2048
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5-Turbo-Raw",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-27",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4-classic": {
        "attachment": true,
        "cost": {
          "input": 27,
          "output": 54
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gpt",
        "id": "openai/gpt-4-classic",
        "last_updated": "2024-03-25",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4-Classic",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-25",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4-classic-0314": {
        "attachment": true,
        "cost": {
          "input": 27,
          "output": 54
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "gpt",
        "id": "openai/gpt-4-classic-0314",
        "last_updated": "2024-08-26",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4-Classic-0314",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-26",
        "status": "deprecated",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 9,
          "output": 27
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo",
        "last_updated": "2023-09-13",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4-Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.45,
          "input": 1.8,
          "output": 7.2
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09,
          "input": 0.36,
          "output": 1.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.022,
          "input": 0.09,
          "output": 0.36
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1-nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "last_updated": "2024-05-13",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4o-aug": {
        "attachment": true,
        "cost": {
          "cache_read": 1.1,
          "input": 2.2,
          "output": 9
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-aug",
        "last_updated": "2024-11-21",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o-Aug",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-21",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.068,
          "input": 0.14,
          "output": 0.54
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 124096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4o-mini-search": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.54
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini-search",
        "last_updated": "2025-03-11",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o-mini-Search",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-4o-search": {
        "attachment": true,
        "cost": {
          "input": 2.2,
          "output": 9
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-4o-search",
        "last_updated": "2025-03-11",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o-Search",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5-chat",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-codex": {
        "attachment": true,
        "cost": {
          "input": 1.1,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5-codex",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-23",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.022,
          "input": 0.22,
          "output": 1.8
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "last_updated": "2025-06-25",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-25",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0045,
          "input": 0.045,
          "output": 0.36
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 14,
          "output": 110
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai/gpt-5-pro",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-06",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "last_updated": "2025-11-12",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-12",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex",
        "last_updated": "2025-11-12",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-12",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.1-codex-max",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex-Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-08",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.022,
          "input": 0.22,
          "output": 1.8
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-mini",
        "last_updated": "2025-11-12",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-12",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1-instant": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-5.1-instant",
        "last_updated": "2025-11-12",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Instant",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-12",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 1.6,
          "output": 13
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5.2",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-08",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 1.6,
          "output": 13
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.2-codex",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-01-14",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-instant": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 1.6,
          "output": 13
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.2-instant",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Instant",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 19,
          "output": 150
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.2-pro",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 1.6,
          "output": 13
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.3-codex",
        "last_updated": "2026-02-10",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-10",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-codex-spark": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.3-codex-spark",
        "last_updated": "2026-03-04",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3-Codex-Spark",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-04",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-instant": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 1.6,
          "output": 13
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.3-instant",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 128000,
          "input": 111616,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3-Instant",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.22,
          "input": 2.2,
          "output": 14
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-5.4",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.068,
          "input": 0.68,
          "output": 4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-mini",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-12",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.018,
          "input": 0.18,
          "output": 1.1
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-nano",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4-Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 27,
          "output": 160
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-5.4-pro",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-5.4-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4545,
          "input": 4.5455,
          "output": 27.2727
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-08",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-08",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "input": 27.2727,
          "output": 163.6364
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "openai/gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-08",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "GPT-5.5-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-08",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-image-1": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-image-1",
        "last_updated": "2025-03-31",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-Image-1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-31",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-image-1-mini": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-image-1-mini",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-Image-1-Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-26",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-image-1.5": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-image-1.5",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "gpt-image-1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": false,
        "tool_call": false
      },
      "openai/gpt-image-2": {
        "attachment": true,
        "cost": {
          "cache_read": 1.2626,
          "input": 5.0505,
          "output": 32.3232
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "openai/gpt-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT-Image-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": false,
        "tool_call": false
      },
      "openai/o1": {
        "attachment": true,
        "cost": {
          "input": 14,
          "output": 54
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "last_updated": "2024-12-18",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-18",
        "temperature": false,
        "tool_call": true
      },
      "openai/o1-pro": {
        "attachment": true,
        "cost": {
          "input": 140,
          "output": 540
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "openai/o1-pro",
        "last_updated": "2025-03-19",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-03-19",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.45,
          "input": 1.8,
          "output": 7.2
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o3",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 2.2,
          "input": 9,
          "output": 36
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o",
        "id": "openai/o3-deep-research",
        "last_updated": "2025-06-27",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-27",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": true,
        "cost": {
          "input": 0.99,
          "output": 4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-01-31",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-mini-high": {
        "attachment": true,
        "cost": {
          "input": 0.99,
          "output": 4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o3-mini-high",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini-high",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-pro": {
        "attachment": true,
        "cost": {
          "input": 18,
          "output": 72
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-pro",
        "id": "openai/o3-pro",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 0.99,
          "output": 4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 0.45,
          "input": 1.8,
          "output": 7.2
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o-mini",
        "id": "openai/o4-mini-deep-research",
        "last_updated": "2025-06-27",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-27",
        "temperature": false,
        "tool_call": true
      },
      "openai/sora-2": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "sora",
        "id": "openai/sora-2",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Sora-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-06",
        "temperature": false,
        "tool_call": true
      },
      "openai/sora-2-pro": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "sora",
        "id": "openai/sora-2-pro",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Sora-2-Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-06",
        "temperature": false,
        "tool_call": true
      },
      "poetools/claude-code": {
        "attachment": true,
        "description": "Claude model for careful reasoning, writing, coding, and tool use",
        "id": "poetools/claude-code",
        "last_updated": "2025-11-27",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "claude-code",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-27",
        "temperature": false,
        "tool_call": true
      },
      "runwayml/runway": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "runway",
        "id": "runwayml/runway",
        "last_updated": "2024-10-11",
        "limit": {
          "context": 256,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Runway",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-11",
        "temperature": false,
        "tool_call": true
      },
      "runwayml/runway-gen-4-turbo": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "runway",
        "id": "runwayml/runway-gen-4-turbo",
        "last_updated": "2025-05-09",
        "limit": {
          "context": 256,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Runway-Gen-4-Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-09",
        "temperature": false,
        "tool_call": true
      },
      "stabilityai/stablediffusionxl": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "stable-diffusion",
        "id": "stabilityai/stablediffusionxl",
        "last_updated": "2023-07-09",
        "limit": {
          "context": 200,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "StableDiffusionXL",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-07-09",
        "temperature": false,
        "tool_call": true
      },
      "topazlabs-co/topazlabs": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "topazlabs",
        "id": "topazlabs-co/topazlabs",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 204,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "TopazLabs",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": false,
        "tool_call": true
      },
      "trytako/tako": {
        "attachment": true,
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "tako",
        "id": "trytako/tako",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 2048,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tako",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-3",
        "last_updated": "2025-04-11",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-11",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-3-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-3-mini",
        "last_updated": "2025-04-11",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 3 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-11",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4",
        "last_updated": "2025-07-10",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok-4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-10",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4-fast-non-reasoning",
        "last_updated": "2025-09-16",
        "limit": {
          "context": 2000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok-4-Fast-Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-16",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4-fast-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4-fast-reasoning",
        "last_updated": "2025-09-16",
        "limit": {
          "context": 2000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok-4-Fast-Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-16",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4.1-fast-non-reasoning": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4.1-fast-non-reasoning",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok-4.1-Fast-Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-19",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4.1-fast-reasoning": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4.1-fast-reasoning",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok-4.1-Fast-Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-19",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-4.20-multi-agent": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 6
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "xai/grok-4.20-multi-agent",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 128000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok-4.20-Multi-Agent",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-13",
        "temperature": false,
        "tool_call": true
      },
      "xai/grok-code-fast-1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-code-fast-1",
        "last_updated": "2025-08-22",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Code Fast 1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-22",
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Poe",
    "npm": "@ai-sdk/openai-compatible"
  },
  "poolside": {
    "api": "https://inference.poolside.ai/v1",
    "doc": "https://platform.poolside.ai",
    "env": [
      "POOLSIDE_API_KEY"
    ],
    "id": "poolside",
    "models": {
      "poolside/laguna-m.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "poolside/laguna-m.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna M.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-28",
        "temperature": true,
        "tool_call": true
      },
      "poolside/laguna-xs.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "poolside/laguna-xs.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Laguna XS.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-28",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Poolside",
    "npm": "@ai-sdk/openai-compatible"
  },
  "privatemode-ai": {
    "api": "http://localhost:8080/v1",
    "doc": "https://docs.privatemode.ai/api/overview",
    "env": [
      "PRIVATEMODE_API_KEY",
      "PRIVATEMODE_ENDPOINT"
    ],
    "id": "privatemode-ai",
    "models": {
      "gemma-3-27b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-3-27b",
        "knowledge": "2024-08",
        "last_updated": "2025-03-12",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "knowledge": "2025-08",
        "last_updated": "2025-08-14",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-embedding-4b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "qwen3-embedding-4b",
        "knowledge": "2025-06",
        "last_updated": "2025-06-06",
        "limit": {
          "context": 32000,
          "output": 2560
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Embedding 4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "whisper-large-v3": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "whisper-large-v3",
        "knowledge": "2023-09",
        "last_updated": "2023-09-01",
        "limit": {
          "context": 0,
          "output": 4096
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper large-v3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-09-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "Privatemode AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "qihang-ai": {
    "api": "https://api.qhaigc.net/v1",
    "doc": "https://www.qhaigc.net/docs",
    "env": [
      "QIHANG_API_KEY"
    ],
    "id": "qihang-ai",
    "models": {
      "claude-haiku-4-5-20251001": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.71
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5-20251001",
        "knowledge": "2025-07-31",
        "last_updated": "2025-10-01",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-5-20251101": {
        "attachment": true,
        "cost": {
          "input": 0.71,
          "output": 3.57
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5-20251101",
        "knowledge": "2025-03",
        "last_updated": "2025-11-01",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-01",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-5-20250929": {
        "attachment": true,
        "cost": {
          "input": 0.43,
          "output": 2.14
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5-20250929",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 0.09,
            "output": 0.71
          },
          "input": 0.09,
          "output": 0.71,
          "tiers": [
            {
              "input": 0.09,
              "output": 0.71,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 0.07,
            "output": 0.43
          },
          "input": 0.07,
          "output": 0.43,
          "tiers": [
            {
              "input": 0.07,
              "output": 0.43,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "input": 0.57,
          "output": 3.43
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3-pro-preview",
        "knowledge": "2025-11",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 1000000,
          "output": 65000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-19",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "input": 0.04,
          "output": 0.29
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-15",
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.2": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 1.14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "QiHang",
    "npm": "@ai-sdk/openai-compatible"
  },
  "qiniu-ai": {
    "api": "https://api.qnaigc.com/v1",
    "doc": "https://developer.qiniu.com/aitokenapi",
    "env": [
      "QINIU_API_KEY"
    ],
    "id": "qiniu-ai",
    "models": {
      "MiniMax-M1": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "MiniMax-M1",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1000000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-3.5-haiku": {
        "attachment": true,
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "claude-3.5-haiku",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-26",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-3.5-sonnet": {
        "attachment": true,
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "claude-3.5-sonnet",
        "last_updated": "2025-09-09",
        "limit": {
          "context": 200000,
          "output": 8200
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-09",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-3.7-sonnet": {
        "attachment": true,
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "claude-3.7-sonnet",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.7 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-4.0-opus": {
        "attachment": true,
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "claude-4.0-opus",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.0 Opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-4.0-sonnet": {
        "attachment": true,
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "claude-4.0-sonnet",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.0 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-4.1-opus": {
        "attachment": true,
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "claude-4.1-opus",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.1 Opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-4.5-haiku": {
        "attachment": true,
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "claude-4.5-haiku",
        "last_updated": "2025-10-16",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.5 Haiku",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-16",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-4.5-opus": {
        "attachment": true,
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "claude-4.5-opus",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.5 Opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-25",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "claude-4.5-sonnet": {
        "attachment": true,
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "claude-4.5-sonnet",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 4.5 Sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1": {
        "attachment": false,
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek-r1",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1-0528": {
        "attachment": false,
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "id": "deepseek-r1-0528",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1-0528",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek-v3",
        "last_updated": "2025-08-13",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-13",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "deepseek-v3-0324": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek-v3-0324",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3-0324",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v3.1": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek-v3.1",
        "last_updated": "2025-08-19",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-math-v2": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-math-v2",
        "last_updated": "2025-12-04",
        "limit": {
          "context": 160000,
          "output": 160000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek/Deepseek-Math-V2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v3.1-terminus": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.1-terminus",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek/DeepSeek-V3.1-Terminus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1-terminus-thinking": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.1-terminus-thinking",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek/DeepSeek-V3.1-Terminus-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v3.2-251201": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-251201",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Deepseek/DeepSeek-V3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-exp": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-exp",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek/DeepSeek-V3.2-Exp",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-29",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-exp-thinking": {
        "attachment": false,
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-exp-thinking",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek/DeepSeek-V3.2-Exp-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-29",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "doubao-1.5-pro-32k": {
        "attachment": false,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "doubao-1.5-pro-32k",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 12000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Pro 32k",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-1.5-thinking-pro": {
        "attachment": false,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "doubao-1.5-thinking-pro",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Thinking Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-1.5-vision-pro": {
        "attachment": true,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "doubao-1.5-vision-pro",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao 1.5 Vision Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "doubao-seed-1.6": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-1.6",
        "last_updated": "2025-08-15",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed 1.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-1.6-flash": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-1.6-flash",
        "last_updated": "2025-08-15",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed 1.6 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-1.6-thinking": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-1.6-thinking",
        "last_updated": "2025-08-15",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed 1.6 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2.0-code": {
        "attachment": true,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "doubao-seed-2.0-code",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Code",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2.0-lite": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-2.0-lite",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2.0-mini": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-2.0-mini",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "doubao-seed-2.0-pro": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "doubao-seed-2.0-pro",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.0-flash": {
        "attachment": true,
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "gemini-2.0-flash",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.0-flash-lite": {
        "attachment": true,
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "gemini-2.0-flash-lite",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1048576,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.0 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "gemini-2.5-flash",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1048576,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-image": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-2.5-flash-image",
        "last_updated": "2025-10-22",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Gemini 2.5 Flash Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "gemini-2.5-flash-lite",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1048576,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "gemini-2.5-pro",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.0-flash-preview": {
        "attachment": true,
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "gemini-3.0-flash-preview",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.0 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-18",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.0-pro-image-preview": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "id": "gemini-3.0-pro-image-preview",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Gemini 3.0 Pro Image Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-20",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "gemini-3.0-pro-preview": {
        "attachment": true,
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "gemini-3.0-pro-preview",
        "last_updated": "2025-11-19",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.0 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "glm-4.5",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-air": {
        "attachment": false,
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "glm-4.5-air",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "id": "gpt-oss-20b",
        "last_updated": "2025-08-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2": {
        "attachment": false,
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "kimi-k2",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "kling-v2-6": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "id": "kling-v2-6",
        "last_updated": "2026-01-13",
        "limit": {
          "context": 99999999,
          "output": 99999999
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling-V2 6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-13",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "meituan/longcat-flash-chat": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "meituan/longcat-flash-chat",
        "last_updated": "2025-11-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meituan/Longcat-Flash-Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "meituan/longcat-flash-lite": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "meituan/longcat-flash-lite",
        "last_updated": "2026-02-06",
        "limit": {
          "context": 256000,
          "output": 320000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meituan/Longcat-Flash-Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "mimo-v2-flash",
        "knowledge": "2024-12-01",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mimo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax/Minimax-M2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-10-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.1",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax/Minimax-M2.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax/Minimax-M2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5-highspeed": {
        "attachment": false,
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "id": "minimax/minimax-m2.5-highspeed",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 204800,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax/Minimax-M2.5 Highspeed",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-14",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-0905": {
        "attachment": false,
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "moonshotai/kimi-k2-0905",
        "last_updated": "2025-09-08",
        "limit": {
          "context": 256000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-08",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "moonshotai/kimi-k2-thinking",
        "last_updated": "2025-11-07",
        "limit": {
          "context": 256000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-07",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "moonshotai/kimi-k2.5",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Moonshotai/Kimi-K2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-28",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": false,
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI/GPT-5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5.2",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI/GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen-max-2025-01-25": {
        "attachment": false,
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "id": "qwen-max-2025-01-25",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen2.5-Max-2025-01-25",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen-turbo": {
        "attachment": false,
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "id": "qwen-turbo",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 1000000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen-vl-max-2025-01-25": {
        "attachment": true,
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen-vl-max-2025-01-25",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen VL-MAX-2025-01-25",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen2.5-vl-72b-instruct": {
        "attachment": true,
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen2.5-vl-72b-instruct",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 VL 72B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen2.5-vl-7b-instruct": {
        "attachment": true,
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen2.5-vl-7b-instruct",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 VL 7B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b": {
        "attachment": false,
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen3-235b-a22b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235B A22B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "qwen3-235b-a22b-instruct-2507",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 262144,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235b A22B Instruct 2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-235b-a22b-thinking-2507",
        "last_updated": "2025-08-12",
        "limit": {
          "context": 262144,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking 2507",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-30b-a3b": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-30b-a3b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 40000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-30b-a3b-instruct-2507": {
        "attachment": false,
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "qwen3-30b-a3b-instruct-2507",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30b A3b Instruct 2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-30b-a3b-thinking-2507": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-30b-a3b-thinking-2507",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 126000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30b A3b Thinking 2507",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-04",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-32b": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-32b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 40000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "qwen3-coder-480b-a35b-instruct",
        "last_updated": "2025-08-14",
        "limit": {
          "context": 262000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-14",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max": {
        "attachment": false,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "qwen3-max",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-24",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-max-preview": {
        "attachment": false,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "qwen3-max-preview",
        "last_updated": "2025-09-06",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "qwen3-next-80b-a3b-instruct",
        "last_updated": "2025-09-12",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "qwen3-next-80b-a3b-thinking",
        "last_updated": "2025-09-12",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-vl-30b-a3b-thinking": {
        "attachment": true,
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "qwen3-vl-30b-a3b-thinking",
        "last_updated": "2026-02-09",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Vl 30b A3b Thinking",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-09",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-397b-a17b": {
        "attachment": true,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "qwen3.5-397b-a17b",
        "last_updated": "2026-02-22",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-22",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/gelab-zero-4b-preview": {
        "attachment": true,
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun-ai/gelab-zero-4b-preview",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Stepfun-Ai/Gelab Zero 4b Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.5-flash": {
        "attachment": true,
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun/step-3.5-flash",
        "last_updated": "2026-02-02",
        "limit": {
          "context": 64000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Stepfun/Step-3.5 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-02",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "x-ai/grok-4-fast": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4-fast",
        "last_updated": "2025-09-20",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "x-AI/Grok-4-Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-20",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4-fast-non-reasoning": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4-fast-non-reasoning",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "X-Ai/Grok-4-Fast-Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-18",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4-fast-reasoning": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4-fast-reasoning",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "X-Ai/Grok-4-Fast-Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-18",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.1-fast": {
        "attachment": false,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.1-fast",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "x-AI/Grok-4.1-Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-20",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.1-fast-non-reasoning": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.1-fast-non-reasoning",
        "last_updated": "2025-12-19",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "X-Ai/Grok 4.1 Fast Non Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.1-fast-reasoning": {
        "attachment": true,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.1-fast-reasoning",
        "last_updated": "2025-12-19",
        "limit": {
          "context": 20000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "X-Ai/Grok 4.1 Fast Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-19",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-code-fast-1": {
        "attachment": false,
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-code-fast-1",
        "last_updated": "2025-09-02",
        "limit": {
          "context": 256000,
          "output": 10000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "x-AI/Grok-Code-Fast 1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-02",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash",
        "knowledge": "2024-12-01",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xiaomi/Mimo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/autoglm-phone-9b": {
        "attachment": true,
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/autoglm-phone-9b",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 12800,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z-Ai/Autoglm Phone 9b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.6",
        "last_updated": "2025-10-11",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z-AI/GLM 4.6",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.7",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z-Ai/GLM 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-23",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5": {
        "attachment": false,
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z-Ai/GLM 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Qiniu",
    "npm": "@ai-sdk/openai-compatible"
  },
  "regolo-ai": {
    "api": "https://api.regolo.ai/v1",
    "doc": "https://docs.regolo.ai/",
    "env": [
      "REGOLO_API_KEY"
    ],
    "id": "regolo-ai",
    "models": {
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 4.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS-120B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 1.8
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-20b",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS-20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-01",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.1-8b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.25
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.1-8b-instruct",
        "last_updated": "2025-04-07",
        "limit": {
          "context": 120000,
          "output": 120000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-07",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 2.7
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "last_updated": "2025-04-28",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 3.5
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.5",
        "last_updated": "2026-03-10",
        "limit": {
          "context": 190000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax 2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-4-119b": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-4-119b",
        "last_updated": "2026-03-15",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 4 119B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-15",
        "temperature": true,
        "tool_call": true
      },
      "mistral-small3.2": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.2
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small3.2",
        "last_updated": "2025-01-31",
        "limit": {
          "context": 120000,
          "output": 120000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-31",
        "temperature": true,
        "tool_call": true
      },
      "qwen-image": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "qwen",
        "id": "qwen-image",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Qwen-Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-01",
        "temperature": true,
        "tool_call": false
      },
      "qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-next",
        "last_updated": "2026-03-01",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-Next",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-embedding-8b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "qwen3-embedding-8b",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Embedding-8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-01",
        "temperature": false,
        "tool_call": false
      },
      "qwen3-reranker-4b": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.12
        },
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "qwen",
        "id": "qwen3-reranker-4b",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 32768,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Reranker-4B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-01",
        "temperature": false,
        "tool_call": false
      },
      "qwen3.5-122b": {
        "attachment": true,
        "cost": {
          "input": 0.9,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-122b",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-122B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3.5-9b": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3.5-9b",
        "last_updated": "2026-02-01",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Regolo AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "requesty": {
    "api": "https://router.requesty.ai/v1",
    "doc": "https://requesty.ai/solution/llm-routing/models",
    "env": [
      "REQUESTY_API_KEY"
    ],
    "id": "requesty",
    "models": {
      "anthropic/claude-3-7-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-3-7-sonnet",
        "knowledge": "2024-01",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 3.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-haiku-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4-5",
        "knowledge": "2025-02-01",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 62000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "context_over_200k": {
            "cache_read": 1,
            "cache_write": 12.5,
            "input": 10,
            "output": 37.5
          },
          "input": 5,
          "output": 25,
          "tiers": [
            {
              "cache_read": 1,
              "cache_write": 12.5,
              "input": 10,
              "output": 37.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "cache_write": 0.55,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.31,
          "cache_write": 2.375,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 1,
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3-flash-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 4.5,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2024-10",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.13,
          "input": 1.25,
          "output": 10
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "audio",
            "image",
            "video"
          ],
          "output": [
            "text",
            "audio",
            "image"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat": {
        "attachment": true,
        "cost": {
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5-chat",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Chat (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-5-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5-codex",
        "knowledge": "2024-10-01",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-image": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-5-image",
        "knowledge": "2024-10-01",
        "last_updated": "2025-10-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "GPT-5 Image",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 16000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt-pro",
        "id": "openai/gpt-5-pro",
        "knowledge": "2024-09-30",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai/gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-chat",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 1.1,
          "output": 9
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex-Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.1-codex-mini",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex-Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-chat",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai/gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "context_over_200k": {
            "cache_read": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 30,
          "input": 30,
          "output": 180
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt-pro",
        "id": "openai/gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.28,
          "input": 1.1,
          "output": 4.4
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-06",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-16",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "cache_write": 3,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4",
        "knowledge": "2025-01",
        "last_updated": "2025-09-09",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-09",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.2,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4-fast",
        "knowledge": "2025-01",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 2000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Requesty",
    "npm": "@ai-sdk/openai-compatible"
  },
  "routing-run": {
    "api": "https://ai.routing.sh/v1",
    "doc": "https://docs.routing.run/api-reference/models",
    "env": [
      "ROUTING_RUN_API_KEY"
    ],
    "id": "routing-run",
    "models": {
      "route/deepseek-v3.2": {
        "attachment": true,
        "cost": {
          "input": 0.4928,
          "output": 0.7392
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "deepseek",
        "id": "route/deepseek-v3.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.4928,
          "output": 0.7392
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "route/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/deepseek-v4-flash-6bit": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.4928,
          "output": 0.7392
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "route/deepseek-v4-flash-6bit",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash 6bit",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.4928,
          "output": 0.7392
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "route/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/deepseek-v4-pro-6bit": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.4928,
          "output": 0.7392
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "route/deepseek-v4-pro-6bit",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro 6bit",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "route/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1,
          "output": 3
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "route/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/glm-5.1-6bit": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1,
          "output": 3
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "route/glm-5.1-6bit",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202752,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1 6bit",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.462,
          "output": 2.42
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "route/kimi-k2.5",
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "route/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.462,
          "output": 2.42
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "route/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/kimi-k2.6-6bit": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.462,
          "output": 2.42
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "route/kimi-k2.6-6bit",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6 6bit",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 0.45,
          "output": 1.35,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "route/mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1000000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "route/mimo-v2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 0.45,
          "output": 1.35,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "route/mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1000000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "route/mimo-v2.5-pro-6bit": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 0.45,
          "output": 1.35,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "route/mimo-v2.5-pro-6bit",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1000000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5 Pro 6bit",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "route/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.193,
          "output": 1.238
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "route/minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 100000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "route/minimax-m2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.193,
          "output": 1.238
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "route/minimax-m2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 100000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5 Highspeed",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "route/minimax-m2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.33,
          "output": 1.32
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "route/minimax-m2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 100000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "route/minimax-m2.7-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.33,
          "output": 1.32
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "route/minimax-m2.7-highspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-18",
        "limit": {
          "context": 100000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7 Highspeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "route/mistral-large-3": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
        "family": "mistral-large",
        "id": "route/mistral-large-3",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "route/mistral-medium-2505": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "route/mistral-medium-2505",
        "knowledge": "2025-05",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 2505",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "route/mistral-small-2503": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents",
        "family": "mistral-small",
        "id": "route/mistral-small-2503",
        "knowledge": "2025-06",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 2503",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": true
      },
      "route/qwen3.6-27b": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 3.3
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "route/qwen3.6-27b",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 202000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/qwen3.6-27b-202k": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 3.3
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "route/qwen3.6-27b-202k",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 202000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B 202K",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "route/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.019,
          "input": 0.096,
          "output": 0.288
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "id": "route/step-3.5-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 262144,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      },
      "route/step-3.5-flash-2603": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "route/step-3.5-flash-2603",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash 2603",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "route/stepfun-3.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.019,
          "input": 0.096,
          "output": 0.288
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "id": "route/stepfun-3.5-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 262144,
          "input": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "StepFun 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "routing.run",
    "npm": "@ai-sdk/openai-compatible"
  },
  "sakana": {
    "api": "https://api.sakana.ai/v1",
    "doc": "https://console.sakana.ai/models",
    "env": [
      "SAKANA_API_KEY"
    ],
    "id": "sakana",
    "models": {
      "fugu": {
        "attachment": true,
        "description": "Multi-agent model for routing expert agents across complex analytical tasks",
        "family": "fugu",
        "id": "fugu",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fugu",
        "open_weights": false,
        "provider": {
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "fugu-ultra": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Quality-first multi-agent model for hard research, analysis, and competitions",
        "family": "fugu",
        "id": "fugu-ultra",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fugu Ultra",
        "open_weights": false,
        "provider": {
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "fugu-ultra-20260615": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Quality-first multi-agent model for hard research, analysis, and competitions",
        "family": "fugu",
        "id": "fugu-ultra-20260615",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fugu Ultra",
        "open_weights": false,
        "provider": {
          "shape": "responses"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Sakana AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "sap-ai-core": {
    "doc": "https://help.sap.com/docs/sap-ai-core",
    "env": [
      "AICORE_SERVICE_KEY"
    ],
    "id": "sap-ai-core",
    "models": {
      "anthropic--claude-3-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic--claude-3-haiku",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-3-haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-3-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic--claude-3-opus",
        "knowledge": "2023-08-31",
        "last_updated": "2024-02-29",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-3-opus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-02-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-3-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic--claude-3-sonnet",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-04",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-3-sonnet",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-04",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-3.5-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic--claude-3.5-sonnet",
        "knowledge": "2024-04-30",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-3.5-sonnet",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-3.7-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic--claude-3.7-sonnet",
        "knowledge": "2024-10-31",
        "last_updated": "2025-02-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-3.7-sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-02-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic--claude-4-opus",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4-opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic--claude-4-sonnet",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4-sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4.5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic--claude-4.5-haiku",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4.5-haiku",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4.5-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic--claude-4.5-opus",
        "knowledge": "2025-05",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4.5-opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4.5-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic--claude-4.5-sonnet",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4.5-sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4.6-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic--claude-4.6-opus",
        "knowledge": "2025-05",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4.6-opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "max"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4.6-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic--claude-4.6-sonnet",
        "knowledge": "2025-08",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4.6-sonnet",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "anthropic--claude-4.7-opus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic--claude-4.7-opus",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "anthropic--claude-4.7-opus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "input_audio": 0.3,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-flash-lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gemini-2.5-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-03-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-4.1-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5-nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "sonar": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar",
        "id": "sonar",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "sonar-deep-research": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 8,
          "reasoning": 3
        },
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar-deep-research",
        "id": "sonar-deep-research",
        "knowledge": "2025-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "sonar-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-02-01",
        "temperature": false,
        "tool_call": false
      },
      "sonar-pro": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "family": "sonar-pro",
        "id": "sonar-pro",
        "knowledge": "2025-09-01",
        "last_updated": "2025-09-01",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "sonar-pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      }
    },
    "name": "SAP AI Core",
    "npm": "@jerome-benoit/sap-ai-provider-v2"
  },
  "sarvam": {
    "api": "https://api.sarvam.ai/v1",
    "doc": "https://docs.sarvam.ai/api-reference-docs/getting-started/models",
    "env": [
      "SARVAM_API_KEY"
    ],
    "id": "sarvam",
    "models": {
      "sarvam-105b": {
        "attachment": false,
        "description": "Flagship Indian-language reasoning model for enterprise multilingual applications",
        "family": "sarvam",
        "id": "sarvam-105b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-06",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sarvam-105B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              null,
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-18",
        "temperature": true,
        "tool_call": true
      },
      "sarvam-30b": {
        "attachment": false,
        "description": "Efficient Indian-language reasoning model for chat, coding, and multilingual work",
        "family": "sarvam",
        "id": "sarvam-30b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-06",
        "limit": {
          "context": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sarvam-30B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              null,
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-18",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Sarvam AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "scaleway": {
    "api": "https://api.scaleway.ai/v1",
    "doc": "https://www.scaleway.com/en/docs/generative-apis/",
    "env": [
      "SCALEWAY_API_KEY"
    ],
    "id": "scaleway",
    "models": {
      "bge-multilingual-gemma2": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "bge-multilingual-gemma2",
        "last_updated": "2025-06-15",
        "limit": {
          "context": 8191,
          "output": 3072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "BGE Multilingual Gemma2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-26",
        "temperature": false,
        "tool_call": false
      },
      "devstral-2-123b-instruct-2512": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral's coding-agent model for repository work, terminal tasks, and software fixes",
        "family": "devstral",
        "id": "devstral-2-123b-instruct-2512",
        "knowledge": "2025-12",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2 123B Instruct (2512)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-07",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 0.5
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-3-27b-it",
        "knowledge": "2024-12",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 40000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma-3-27B-IT",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2024-12-01",
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 0.5
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-26b-a4b-it",
        "knowledge": "2025-04",
        "last_updated": "2026-05-22",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-01",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "input": 1.8,
          "output": 5.5
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": true
      },
      "llama-3.3-70b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.9,
          "output": 0.9
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b-instruct",
        "knowledge": "2023-12",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 100000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "mistral-medium-3.5-128b": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Balanced Mistral model for enterprise assistants, multilingual work, and tools",
        "family": "mistral-medium",
        "id": "mistral-medium-3.5-128b",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.5 128B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-3.2-24b-instruct-2506": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.35
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-3.2-24b-instruct-2506",
        "knowledge": "2025-03",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2 24B Instruct (2506)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-06-20",
        "temperature": true,
        "tool_call": true
      },
      "pixtral-12b-2409": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Mistral vision-language model for image understanding and multimodal chat",
        "family": "pixtral",
        "id": "pixtral-12b-2409",
        "knowledge": "2024-09",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral 12B 2409",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.75,
          "output": 2.25,
          "reasoning": 8.4
        },
        "description": "Large open Qwen MoE for multilingual reasoning, coding, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b-instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 260000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-30b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Smaller Qwen coder for efficient local agents and repo-level fixes",
        "family": "qwen",
        "id": "qwen3-coder-30b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder 30B-A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04",
        "temperature": true,
        "tool_call": true
      },
      "qwen3-embedding-8b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "qwen3-embedding-8b",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-25-11",
        "temperature": false,
        "tool_call": false
      },
      "qwen3.5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen3.5-397b-a17b",
        "knowledge": "2025-04",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.5
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "qwen3.6-35b-a3b",
        "knowledge": "2025-04",
        "last_updated": "2026-05-22",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-01",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "voxtral-small-24b-2507": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.35
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "voxtral",
        "id": "voxtral-small-24b-2507",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voxtral Small 24B 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-01",
        "temperature": true,
        "tool_call": true
      },
      "whisper-large-v3": {
        "attachment": false,
        "cost": {
          "input": 0.003,
          "output": 0
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "whisper-large-v3",
        "knowledge": "2023-09",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 0,
          "output": 8192
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper Large v3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-09-01",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "Scaleway",
    "npm": "@ai-sdk/openai-compatible"
  },
  "siliconflow": {
    "api": "https://api.siliconflow.com/v1",
    "doc": "https://cloud.siliconflow.com/models",
    "env": [
      "SILICONFLOW_API_KEY"
    ],
    "id": "siliconflow",
    "models": {
      "ByteDance-Seed/Seed-OSS-36B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.21,
          "output": 0.57
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "seed",
        "id": "ByteDance-Seed/Seed-OSS-36B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance-Seed/Seed-OSS-36B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 197000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMaxAI/MiniMax-M2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-15",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-72B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.59,
          "output": 0.59
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-72B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 33000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen2.5-72B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-7B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.05
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-7B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 33000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen2.5-7B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-14B": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-14B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-14B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-32B": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-32B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-8B": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.06
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-8B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-8B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-30B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-235B-A22B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-235B-A22B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-235B-A22B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-235B-A22B-Thinking": {
        "attachment": true,
        "cost": {
          "input": 0.45,
          "output": 3.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-235B-A22B-Thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-235B-A22B-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-30B-A3B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.29,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-30B-A3B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-30B-A3B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-30B-A3B-Thinking": {
        "attachment": true,
        "cost": {
          "input": 0.29,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-30B-A3B-Thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-30B-A3B-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-32B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-32B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-32B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-32B-Thinking": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-32B-Thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-32B-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-8B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.68
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-8B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-8B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-122B-A10B": {
        "attachment": false,
        "cost": {
          "input": 0.26,
          "output": 2.08
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-122B-A10B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 122B-A10B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-27B": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-27B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 27B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-35B-A3B": {
        "attachment": false,
        "cost": {
          "input": 0.24,
          "output": 1.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-35B-A3B",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 35B-A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B": {
        "attachment": false,
        "cost": {
          "input": 0.39,
          "output": 2.34
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B",
        "last_updated": "2026-02-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B-A17B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-9B": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.15
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-9B",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-9B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-27B": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 3.2
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-27B",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.6
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-35B-A3B",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B-A3B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "baidu/ERNIE-4.5-300B-A47B": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.1
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "ernie",
        "id": "baidu/ERNIE-4.5-300B-A47B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "baidu/ERNIE-4.5-300B-A47B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.18
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1-Terminus": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1-Terminus",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3.1-Terminus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.42
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "last_updated": "2025-12-03",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2-Exp": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.41
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2-Exp",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3.2-Exp",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-ai/DeepSeek-V4-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.145,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-26B-A4B-it": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.4
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26B-A4B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31B-it": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.4
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionAI/Ling-flash-2.0": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "ling",
        "id": "inclusionAI/Ling-flash-2.0",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "inclusionAI/Ling-flash-2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 2.25
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "moonshotai/Kimi-K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.77,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-15",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "moonshotai/Kimi-K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.45
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "openai/gpt-oss-120b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.18
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "openai/gpt-oss-20b",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/Step-3.5-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "family": "step",
        "id": "stepfun-ai/Step-3.5-Flash",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "stepfun-ai/Step-3.5-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "tencent/Hunyuan-A13B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "tencent/Hunyuan-A13B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "tencent/Hunyuan-A13B-Instruct",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "tencent/Hy3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.029,
          "input": 0.066,
          "output": 0.26
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "Hy",
        "id": "tencent/Hy3-preview",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3 preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-Air": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.86
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "zai-org/GLM-4.5-Air",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "zai-org/GLM-4.5-Air",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "input": 0.95,
          "output": 2.55
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-15",
        "limit": {
          "context": 205000,
          "output": 205000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "zai-org/GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-08",
        "limit": {
          "context": 205000,
          "output": 205000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "zai-org/GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1049000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5V-Turbo": {
        "attachment": true,
        "cost": {
          "cache_write": 0,
          "input": 1.2,
          "output": 4
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "zai-org/GLM-5V-Turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "zai-org/GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "SiliconFlow",
    "npm": "@ai-sdk/openai-compatible"
  },
  "siliconflow-cn": {
    "api": "https://api.siliconflow.cn/v1",
    "doc": "https://cloud.siliconflow.com/models",
    "env": [
      "SILICONFLOW_CN_API_KEY"
    ],
    "id": "siliconflow-cn",
    "models": {
      "ByteDance-Seed/Seed-OSS-36B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.21,
          "output": 0.57
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "seed",
        "id": "ByteDance-Seed/Seed-OSS-36B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ByteDance-Seed/Seed-OSS-36B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "PaddlePaddle/PaddleOCR-VL-1.5": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "PaddlePaddle/PaddleOCR-VL-1.5",
        "last_updated": "2026-01-29",
        "limit": {
          "context": 16384,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "PaddlePaddle/PaddleOCR-VL-1.5",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": false
      },
      "Pro/MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.22
        },
        "description": "Frontier MiniMax model for engineering, office tasks, and agentic reasoning",
        "family": "minimax",
        "id": "Pro/MiniMaxAI/MiniMax-M2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-13",
        "limit": {
          "context": 192000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/MiniMaxAI/MiniMax-M2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/deepseek-ai/DeepSeek-R1": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.18
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "Pro/deepseek-ai/DeepSeek-R1",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/deepseek-ai/DeepSeek-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/deepseek-ai/DeepSeek-V3": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek",
        "id": "Pro/deepseek-ai/DeepSeek-V3",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/deepseek-ai/DeepSeek-V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/deepseek-ai/DeepSeek-V3.1-Terminus": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek",
        "id": "Pro/deepseek-ai/DeepSeek-V3.1-Terminus",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/deepseek-ai/DeepSeek-V3.1-Terminus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.42
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek",
        "id": "Pro/deepseek-ai/DeepSeek-V3.2",
        "last_updated": "2025-12-03",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/deepseek-ai/DeepSeek-V3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/moonshotai/Kimi-K2.5": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 2.25
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi",
        "id": "Pro/moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/moonshotai/Kimi-K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/moonshotai/Kimi-K2.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi",
        "id": "Pro/moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/moonshotai/Kimi-K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "Pro/zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 205000,
          "output": 205000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/zai-org/GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Pro/zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "Pro/zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-08",
        "limit": {
          "context": 205000,
          "output": 205000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pro/zai-org/GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-08",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-72B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.59,
          "output": 0.59
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-72B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 33000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen2.5-72B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-7B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.05
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-7B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 33000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen2.5-7B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-14B": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-14B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-14B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-07-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.09,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-32B": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-32B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-8B": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.06
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-8B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-8B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-30B-A3B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.28
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-235B-A22B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-235B-A22B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-235B-A22B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-235B-A22B-Thinking": {
        "attachment": true,
        "cost": {
          "input": 0.45,
          "output": 3.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-235B-A22B-Thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-235B-A22B-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-30B-A3B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.29,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-30B-A3B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-30B-A3B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-30B-A3B-Thinking": {
        "attachment": true,
        "cost": {
          "input": 0.29,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-30B-A3B-Thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-30B-A3B-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-32B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-32B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-32B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-32B-Thinking": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.5
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-32B-Thinking",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-32B-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-8B-Instruct": {
        "attachment": true,
        "cost": {
          "input": 0.18,
          "output": 0.68
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-8B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3-VL-8B-Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-122B-A10B": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 2.32
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-122B-A10B",
        "knowledge": "2025-04",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-122B-A10B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-27B": {
        "attachment": false,
        "cost": {
          "input": 0.26,
          "output": 2.09
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-27B",
        "knowledge": "2025-04",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-25",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-35B-A3B": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "output": 1.86
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-35B-A3B",
        "knowledge": "2025-04",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-25",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B": {
        "attachment": false,
        "cost": {
          "input": 0.29,
          "output": 1.74
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-4B": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-4B",
        "knowledge": "2025-04",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-4B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-9B": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 1.74
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-9B",
        "knowledge": "2025-04",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.5-9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-35B-A3B": {
        "attachment": false,
        "cost": {
          "input": 0.23,
          "output": 1.86
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-35B-A3B",
        "knowledge": "2025-04",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen/Qwen3.6-35B-A3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-17",
        "temperature": true,
        "tool_call": true
      },
      "baidu/ERNIE-4.5-300B-A47B": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 1.1
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "family": "ernie",
        "id": "baidu/ERNIE-4.5-300B-A47B",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "baidu/ERNIE-4.5-300B-A47B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-OCR": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "OCR model for extracting structured text from documents and screenshots",
        "id": "deepseek-ai/DeepSeek-OCR",
        "last_updated": "2025-10-20",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-OCR",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-20",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-R1": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.18
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1-Terminus": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1-Terminus",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3.1-Terminus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.2": {
        "attachment": false,
        "cost": {
          "input": 0.27,
          "output": 0.42
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.2",
        "last_updated": "2025-12-03",
        "limit": {
          "context": 164000,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek-ai/DeepSeek-V4-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.145,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1049000,
          "output": 393000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "deepseek-ai/DeepSeek-V4-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "inclusionAI/Ling-flash-2.0": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "ling",
        "id": "inclusionAI/Ling-flash-2.0",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "inclusionAI/Ling-flash-2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "stepfun-ai/Step-3.5-Flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "family": "step",
        "id": "stepfun-ai/Step-3.5-Flash",
        "last_updated": "2026-02-11",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "stepfun-ai/Step-3.5-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "tencent/Hunyuan-A13B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.57
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "tencent/Hunyuan-A13B-Instruct",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "tencent/Hunyuan-A13B-Instruct",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-06-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-Air": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.86
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "zai-org/GLM-4.5-Air",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "zai-org/GLM-4.5-Air",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1049000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "SiliconFlow (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "snowflake-cortex": {
    "api": "https://${SNOWFLAKE_ACCOUNT}.snowflakecomputing.com/api/v2/cortex/v1",
    "doc": "https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api",
    "env": [
      "SNOWFLAKE_ACCOUNT",
      "SNOWFLAKE_CORTEX_PAT"
    ],
    "id": "snowflake-cortex",
    "models": {
      "claude-fable-5": {
        "attachment": true,
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "claude-haiku-4-5": {
        "attachment": true,
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "claude-haiku-4-5",
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 3,
                "cache_write": 37.5,
                "input": 30,
                "output": 150
              },
              "provider": {
                "body": {
                  "speed": "fast"
                },
                "headers": {
                  "anthropic-beta": "fast-mode-2026-02-01"
                }
              }
            }
          }
        },
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "status": "beta",
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5 (latest)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-r1": {
        "attachment": false,
        "description": "Classic open reasoning model for transparent math, coding, and deliberate problem solving",
        "family": "deepseek-thinking",
        "id": "deepseek-r1",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro": {
        "attachment": true,
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-large2": {
        "attachment": true,
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral-large2",
        "knowledge": "2024-11",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-4.1": {
        "attachment": true,
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai-gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-5": {
        "attachment": true,
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai-gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "status": "beta",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5-mini": {
        "attachment": true,
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai-gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "input": 272000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "status": "beta",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5-nano": {
        "attachment": true,
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "openai-gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "status": "beta",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.1": {
        "attachment": true,
        "description": "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai-gpt-5.1",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.2": {
        "attachment": true,
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai-gpt-5.2",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.4": {
        "attachment": true,
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 0.5,
                "input": 5,
                "output": 30
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai-gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "status": "beta",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-5.5": {
        "attachment": true,
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai-gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "status": "beta",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "snowflake-llama3.3-70b": {
        "attachment": true,
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "snowflake-llama3.3-70b",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Snowflake Cortex",
    "npm": "@ai-sdk/openai-compatible"
  },
  "stackit": {
    "api": "https://api.openai-compat.model-serving.eu01.onstackit.cloud/v1",
    "doc": "https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models",
    "env": [
      "STACKIT_API_KEY"
    ],
    "id": "stackit",
    "models": {
      "Qwen/Qwen3-VL-235B-A22B-Instruct-FP8": {
        "attachment": true,
        "cost": {
          "input": 1.64,
          "output": 1.91
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-235B-A22B-Instruct-FP8",
        "last_updated": "2024-11-01",
        "limit": {
          "context": 218000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL 235B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-VL-Embedding-8B": {
        "attachment": true,
        "cost": {
          "input": 0.09,
          "output": 0.09
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "Qwen/Qwen3-VL-Embedding-8B",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 32000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-VL Embedding 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-05",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic": {
        "attachment": false,
        "cost": {
          "input": 0.49,
          "output": 0.71
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.49,
          "output": 0.71
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3-27b-it",
        "last_updated": "2025-05-17",
        "limit": {
          "context": 37000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3 27B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-17",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "intfloat/e5-mistral-7b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.02,
          "output": 0.02
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "mistral",
        "id": "intfloat/e5-mistral-7b-instruct",
        "last_updated": "2023-12-11",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "E5 Mistral 7B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2023-12-11",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      },
      "neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.16,
          "output": 0.27
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "neuralmagic/Meta-Llama-3.1-8B-Instruct-FP8",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "neuralmagic/Mistral-Nemo-Instruct-2407-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.49,
          "output": 0.71
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral",
        "id": "neuralmagic/Mistral-Nemo-Instruct-2407-FP8",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.49,
          "output": 0.71
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "STACKIT",
    "npm": "@ai-sdk/openai-compatible"
  },
  "stepfun": {
    "api": "https://api.stepfun.com/v1",
    "doc": "https://platform.stepfun.com/docs/zh/overview/concept",
    "env": [
      "STEPFUN_API_KEY"
    ],
    "id": "stepfun",
    "models": {
      "step-1-32k": {
        "attachment": false,
        "cost": {
          "cache_read": 0.41,
          "input": 2.05,
          "output": 9.59
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "step-1-32k",
        "knowledge": "2024-06",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 1 (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "step-2-16k": {
        "attachment": false,
        "cost": {
          "cache_read": 1.04,
          "input": 5.21,
          "output": 16.44
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "step-2-16k",
        "knowledge": "2024-06",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 2 (16K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "step-3.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "id": "step-3.5-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      },
      "step-3.5-flash-2603": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "step-3.5-flash-2603",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash 2603",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "step-3.7-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.2,
          "output": 1.15
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "id": "step-3.7-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01-01",
        "last_updated": "2026-06-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": true
      },
      "step-tts-2": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "step",
        "id": "step-tts-2",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Step TTS 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-01",
        "temperature": false,
        "tool_call": false
      },
      "stepaudio-2.5-asr": {
        "attachment": false,
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "step",
        "id": "stepaudio-2.5-asr",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "StepAudio 2.5 ASR",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-24",
        "temperature": false,
        "tool_call": false
      },
      "stepaudio-2.5-tts": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "step",
        "id": "stepaudio-2.5-tts",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "StepAudio 2.5 TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "StepFun",
    "npm": "@ai-sdk/openai-compatible"
  },
  "stepfun-ai": {
    "api": "https://api.stepfun.ai/step_plan/v1",
    "doc": "https://platform.stepfun.ai/docs/en/step-plan/integrations/open-code",
    "env": [
      "STEPFUN_API_KEY"
    ],
    "id": "stepfun-ai",
    "models": {
      "step-1-32k": {
        "attachment": false,
        "cost": {
          "cache_read": 0.41,
          "input": 2.05,
          "output": 9.59
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "step-1-32k",
        "knowledge": "2024-06",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 32768,
          "input": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 1 (32K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "step-2-16k": {
        "attachment": false,
        "cost": {
          "cache_read": 1.04,
          "input": 5.21,
          "output": 16.44
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "step-2-16k",
        "knowledge": "2024-06",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 16384,
          "input": 16384,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 2 (16K)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-01",
        "temperature": true,
        "tool_call": true
      },
      "step-3.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "id": "step-3.5-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      },
      "step-3.5-flash-2603": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "step-3.5-flash-2603",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash 2603",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "step-3.7-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.2,
          "output": 1.15
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "id": "step-3.7-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2026-01-01",
        "last_updated": "2026-06-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": true
      },
      "step-tts-2": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "step",
        "id": "step-tts-2",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Step TTS 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-01",
        "temperature": false,
        "tool_call": false
      },
      "stepaudio-2.5-asr": {
        "attachment": false,
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "step",
        "id": "stepaudio-2.5-asr",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "StepAudio 2.5 ASR",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-24",
        "temperature": false,
        "tool_call": false
      },
      "stepaudio-2.5-tts": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "step",
        "id": "stepaudio-2.5-tts",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "StepAudio 2.5 TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "StepFun AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "subconscious": {
    "api": "https://api.subconscious.dev/v1",
    "doc": "https://docs.subconscious.dev",
    "env": [
      "SUBCONSCIOUS_API_KEY"
    ],
    "id": "subconscious",
    "models": {
      "subconscious/tim-qwen3.6-27b": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "subconscious/tim-qwen3.6-27b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-11",
        "limit": {
          "context": 8192,
          "input": 8192,
          "output": 5000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "TIM-Qwen3.6 27B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Subconscious",
    "npm": "@ai-sdk/openai-compatible"
  },
  "submodel": {
    "api": "https://llm.submodel.ai/v1",
    "doc": "https://submodel.gitbook.io",
    "env": [
      "SUBMODEL_INSTAGEN_ACCESS_KEY"
    ],
    "id": "submodel",
    "models": {
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-R1-0528": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.15
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1-0528",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 75000,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek R1 0528",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3-0324": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3-0324",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 75000,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 75000,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.5
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2025-08-23",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-23",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-Air": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.5
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "zai-org/GLM-4.5-Air",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-4.5-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-4.5-FP8",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 FP8",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "submodel",
    "npm": "@ai-sdk/openai-compatible"
  },
  "synthetic": {
    "api": "https://api.synthetic.new/openai/v1",
    "doc": "https://synthetic.new/pricing",
    "env": [
      "SYNTHETIC_API_KEY"
    ],
    "id": "synthetic",
    "models": {
      "hf:MiniMaxAI/MiniMax-M3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.6,
          "input": 0.6,
          "output": 1.2
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "hf:MiniMaxAI/MiniMax-M3",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-12",
        "limit": {
          "context": 524288,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "hf:Qwen/Qwen3.6-27B": {
        "attachment": true,
        "cost": {
          "cache_read": 0.45,
          "input": 0.45,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "hf:Qwen/Qwen3.6-27B",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-22",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 27B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "hf:moonshotai/Kimi-K2.7-Code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.95,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "hf:moonshotai/Kimi-K2.7-Code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 0.3,
          "output": 1
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-11",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Super 120B A12B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "hf:openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.1,
          "output": 0.1
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "hf:openai/gpt-oss-120b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "hf:zai-org/GLM-4.7-Flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "input": 0.1,
          "output": 0.5
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm-flash",
        "id": "hf:zai-org/GLM-4.7-Flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 196608,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "hf:zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 1.4,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "hf:zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 524288,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Synthetic",
    "npm": "@ai-sdk/openai-compatible"
  },
  "tencent-coding-plan": {
    "api": "https://api.lkeap.cloud.tencent.com/coding/v3",
    "doc": "https://cloud.tencent.com/document/product/1772/128947",
    "env": [
      "TENCENT_CODING_PLAN_API_KEY"
    ],
    "id": "tencent-coding-plan",
    "models": {
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "hunyuan-2.0-instruct": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "hunyuan-2.0-instruct",
        "last_updated": "2026-03-08",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tencent HY 2.0 Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-08",
        "temperature": true,
        "tool_call": true
      },
      "hunyuan-2.0-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "hunyuan-2.0-thinking",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-08",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Tencent HY 2.0 Think",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-08",
        "temperature": true,
        "tool_call": true
      },
      "hunyuan-t1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "hunyuan-t1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-08",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hunyuan-T1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-08",
        "temperature": true,
        "tool_call": true
      },
      "hunyuan-turbos": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "hunyuan",
        "id": "hunyuan-turbos",
        "last_updated": "2026-03-08",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hunyuan-TurboS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-08",
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "minimax-m2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "tc-code-latest": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Automatic model router for matching prompts to suitable backends and budgets",
        "family": "auto",
        "id": "tc-code-latest",
        "last_updated": "2026-03-08",
        "limit": {
          "context": 131072,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-08",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Tencent Coding Plan (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "tencent-token-plan": {
    "api": "https://api.lkeap.cloud.tencent.com/plan/v3",
    "doc": "https://cloud.tencent.com/document/product/1823/130060",
    "env": [
      "TENCENT_TOKEN_PLAN_API_KEY"
    ],
    "id": "tencent-token-plan",
    "models": {
      "hy3": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "Hy",
        "id": "hy3",
        "last_updated": "2026-07-06",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-07-06",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Tencent Token Plan",
    "npm": "@ai-sdk/openai-compatible"
  },
  "tencent-tokenhub": {
    "api": "https://tokenhub.tencentmaas.com/v1",
    "doc": "https://cloud.tencent.com/document/product/1823/130050",
    "env": [
      "TENCENT_TOKENHUB_API_KEY"
    ],
    "id": "tencent-tokenhub",
    "models": {
      "hy3": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "Hy",
        "id": "hy3",
        "last_updated": "2026-07-06",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-07-06",
        "temperature": true,
        "tool_call": true
      },
      "hy3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "Hy",
        "id": "hy3-preview",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3 preview",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Tencent TokenHub",
    "npm": "@ai-sdk/openai-compatible"
  },
  "the-grid-ai": {
    "api": "https://api.thegrid.ai/v1",
    "doc": "https://thegrid.ai/docs",
    "env": [
      "THEGRIDAI_API_KEY"
    ],
    "id": "the-grid-ai",
    "models": {
      "agent-max": {
        "attachment": false,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "agent-max",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agent Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-04",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "agent-prime": {
        "attachment": false,
        "description": "Preview model for early access evaluation, prototyping, and compatibility testing",
        "id": "agent-prime",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agent Prime",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-04",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "agent-standard": {
        "attachment": false,
        "description": "Preview model for early access evaluation, prototyping, and compatibility testing",
        "id": "agent-standard",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agent Standard",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-04",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "code-max": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "code-max",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Code Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-04",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "code-prime": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "code-prime",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Code Prime",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-04",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "code-standard": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "code-standard",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Code Standard",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-04",
        "status": "beta",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "text-max": {
        "attachment": false,
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "text-max",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Text Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "text-prime": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "text-prime",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 128000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Text Prime",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "text-standard": {
        "attachment": false,
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "text-standard",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 128000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Text Standard",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "The Grid AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "tinfoil": {
    "api": "https://inference.tinfoil.sh/v1",
    "doc": "https://docs.tinfoil.sh",
    "env": [
      "TINFOIL_API_KEY"
    ],
    "id": "tinfoil",
    "models": {
      "gemma4-31b": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "gemma4-31b",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5-2": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 5.25
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5-2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 384000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "gpt-oss-120b",
        "knowledge": "2024-06",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-oss-safeguard-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "gpt-oss-safeguard-120b",
        "knowledge": "2024-06",
        "last_updated": "2025-10-29",
        "limit": {
          "context": 131000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-safeguard-120b",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-6": {
        "attachment": true,
        "cost": {
          "input": 1.5,
          "output": 5.25
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2-6",
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 256000,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "llama3-3-70b": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 2.75
        },
        "description": "Popular open Llama workhorse for multilingual chat, coding, and self-hosting",
        "family": "llama",
        "id": "llama3-3-70b",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "nomic-embed-text": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0
        },
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "id": "nomic-embed-text",
        "last_updated": "2024-02",
        "limit": {
          "context": 8192,
          "output": 768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nomic Embed Text v1.5",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-02",
        "structured_output": false,
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "Tinfoil",
    "npm": "@ai-sdk/openai-compatible"
  },
  "togetherai": {
    "doc": "https://docs.together.ai/docs/serverless-models",
    "env": [
      "TOGETHER_API_KEY"
    ],
    "id": "togetherai",
    "models": {
      "LiquidAI/LFM2-24B-A2B": {
        "attachment": false,
        "cost": {
          "input": 0.03,
          "output": 0.12
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "liquid",
        "id": "LiquidAI/LFM2-24B-A2B",
        "last_updated": "2026-02-25",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LFM2-24B-A2B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-25",
        "temperature": true,
        "tool_call": false
      },
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "MiniMaxAI/MiniMax-M3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M3",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 524288,
          "output": 250000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen2.5-7B-Instruct-Turbo": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.3
        },
        "description": "Efficient Qwen model for fast chat, extraction, and high-volume workloads",
        "family": "qwen",
        "id": "Qwen/Qwen2.5-7B-Instruct-Turbo",
        "last_updated": "2024-09-19",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 2.5 7B Instruct Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Instruct-2507-tput": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507-tput",
        "knowledge": "2025-07",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507 FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-Next-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.2
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-Next-FP8",
        "knowledge": "2026-02-03",
        "last_updated": "2026-02-03",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-03",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-397B-A17B": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-397B-A17B",
        "last_updated": "2026-06-15",
        "limit": {
          "context": 262144,
          "output": 130000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 397B A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.5-9B": {
        "attachment": true,
        "cost": {
          "input": 0.17,
          "output": 0.25
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen/Qwen3.5-9B",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.6-Plus": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3.6-Plus",
        "last_updated": "2026-04-30",
        "limit": {
          "context": 1000000,
          "output": 500000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 Plus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-30",
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3.7-Max": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 3.75
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "Qwen/Qwen3.7-Max",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 1000000,
          "output": 500000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "deepcogito/cogito-v2-1-671b": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 1.25
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "cogito",
        "id": "deepcogito/cogito-v2-1-671b",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cogito v2.1 671B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-11-13",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-R1": {
        "attachment": false,
        "cost": {
          "input": 3,
          "output": 7
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "deepseek-thinking",
        "id": "deepseek-ai/DeepSeek-R1",
        "knowledge": "2024-07",
        "last_updated": "2025-03-24",
        "limit": {
          "context": 163839,
          "output": 163839
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "status": "deprecated",
        "temperature": true,
        "tool_call": false
      },
      "deepseek-ai/DeepSeek-V3": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 1.25
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-26",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3-1": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 1.7
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3-1",
        "knowledge": "2025-08",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-21",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V4-Pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V4-Pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-24",
        "limit": {
          "context": 512000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "essentialai/Rnj-1-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
        "family": "rnj",
        "id": "essentialai/Rnj-1-Instruct",
        "knowledge": "2024-10",
        "last_updated": "2025-12-05",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Rnj-1 Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-12-05",
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-3n-E4B-it": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.12
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-3n-E4B-it",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 3N E4B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-05-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-4-31B-it": {
        "attachment": true,
        "cost": {
          "input": 0.39,
          "output": 0.97
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-31B-it",
        "knowledge": "2025-01",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
        "attachment": false,
        "cost": {
          "input": 1.04,
          "output": 1.04
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
        "knowledge": "2023-12",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Meta-Llama-3-8B-Instruct-Lite": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.14
        },
        "description": "Compact Llama instruction model for fast chat and local deployment",
        "family": "llama",
        "id": "meta-llama/Meta-Llama-3-8B-Instruct-Lite",
        "last_updated": "2024-04-18",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta Llama 3 8B Instruct Lite",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-04-18",
        "temperature": true,
        "tool_call": false
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 2.8
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": true,
        "knowledge": "2026-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.2,
          "output": 4.5
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.7-Code": {
        "attachment": false,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi coding model for software agents, refactors, and repository reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.7-Code",
        "last_updated": "2026-06-14",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-3-ultra-550b-a55b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.6,
          "output": 3.6
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-ultra-550b-a55b",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 512300,
          "output": 512300
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra 550B A55B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "knowledge": "2025-08",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "pearl-ai/gemma-4-31b-it": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 0.86
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "pearl-ai/gemma-4-31b-it",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pearl AI Gemma 4 31B Instruct",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "temperature": true,
        "tool_call": false
      },
      "zai-org/GLM-5": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3.2
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "zai-org/GLM-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "input": 1.4,
          "output": 4.4
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "knowledge": "2025-11",
        "last_updated": "2026-07-02",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org/GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-16",
        "limit": {
          "context": 262144,
          "output": 164000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Together AI",
    "npm": "@ai-sdk/togetherai"
  },
  "trustedrouter": {
    "api": "https://api.trustedrouter.com/v1",
    "doc": "https://trustedrouter.com/docs",
    "env": [
      "TRUSTEDROUTER_API_KEY"
    ],
    "id": "trustedrouter",
    "models": {
      "auto": {
        "attachment": true,
        "description": "TrustedRouter automatic routing alias that chooses a healthy supported model endpoint for the request.",
        "id": "auto",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Auto",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "cheap": {
        "attachment": true,
        "description": "TrustedRouter low-cost routing alias that prefers inexpensive healthy model endpoints.",
        "id": "cheap",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cheap",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "e2e": {
        "attachment": true,
        "description": "TrustedRouter privacy routing alias for end-to-end encrypted provider routes where available.",
        "id": "e2e",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "End-to-End Encrypted",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "fast": {
        "attachment": true,
        "description": "TrustedRouter speed routing alias that prefers low-latency healthy model endpoints.",
        "id": "fast",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "synth": {
        "attachment": true,
        "description": "TrustedRouter synthesis orchestration alias that combines multiple model responses into one answer.",
        "id": "synth",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Synth",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "synth-code": {
        "attachment": true,
        "description": "TrustedRouter code synthesis orchestration alias that combines multiple model responses into one answer.",
        "id": "synth-code",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Synth Code",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zdr": {
        "attachment": true,
        "description": "TrustedRouter privacy routing alias that prefers zero data retention model endpoints.",
        "id": "zdr",
        "last_updated": "2026-06-27",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Zero Data Retention",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "TrustedRouter",
    "npm": "@ai-sdk/openai-compatible"
  },
  "umans-ai": {
    "api": "https://api.code.umans.ai/v1",
    "doc": "https://app.umans.ai/offers/code/docs/orgs",
    "env": [
      "UMANS_AI_API_KEY"
    ],
    "id": "umans-ai",
    "models": {
      "umans-coder": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "umans-coder",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Umans Coder",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "umans-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.15,
          "output": 1
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "umans-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Umans Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "umans-glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.29,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "umans-glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "umans-glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "umans-glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 405504,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "umans-kimi-k2.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "umans-kimi-k2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Umans AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "umans-ai-coding-plan": {
    "api": "https://api.code.umans.ai/v1",
    "doc": "https://app.umans.ai/offers/code/docs",
    "env": [
      "UMANS_AI_CODING_PLAN_API_KEY"
    ],
    "id": "umans-ai-coding-plan",
    "models": {
      "umans-coder": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "umans-coder",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Umans Coder",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "umans-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "umans-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Umans Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "umans-glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "umans-glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "umans-glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "umans-glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 405504,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "umans-kimi-k2.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "umans-kimi-k2.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "umans-qwen3.6-35b-a3b": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Qwen MoE for local agents that need vision, audio, and code",
        "family": "qwen",
        "id": "umans-qwen3.6-35b-a3b",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-17",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Umans AI Coding Plan",
    "npm": "@ai-sdk/openai-compatible"
  },
  "upstage": {
    "api": "https://api.upstage.ai/v1/solar",
    "doc": "https://developers.upstage.ai/docs/apis/chat",
    "env": [
      "UPSTAGE_API_KEY"
    ],
    "id": "upstage",
    "models": {
      "solar-mini": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "solar-mini",
        "id": "solar-mini",
        "knowledge": "2024-09",
        "last_updated": "2025-04-22",
        "limit": {
          "context": 32768,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "solar-mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-12",
        "temperature": true,
        "tool_call": true
      },
      "solar-pro2": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.25
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "solar-pro",
        "id": "solar-pro2",
        "knowledge": "2025-03",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 65536,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "solar-pro2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": true
      },
      "solar-pro3": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.25
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "solar-pro",
        "id": "solar-pro3",
        "knowledge": "2025-03",
        "last_updated": "2026-01",
        "limit": {
          "context": 131072,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "solar-pro3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Upstage",
    "npm": "@ai-sdk/openai-compatible"
  },
  "v0": {
    "doc": "https://sdk.vercel.ai/providers/ai-sdk-providers/vercel",
    "env": [
      "V0_API_KEY"
    ],
    "id": "v0",
    "models": {
      "v0-1.0-md": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "v0",
        "id": "v0-1.0-md",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "v0-1.0-md",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "v0-1.5-lg": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 75
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "v0",
        "id": "v0-1.5-lg",
        "last_updated": "2025-06-09",
        "limit": {
          "context": 512000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "v0-1.5-lg",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-09",
        "temperature": true,
        "tool_call": true
      },
      "v0-1.5-md": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 15
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "v0",
        "id": "v0-1.5-md",
        "last_updated": "2025-06-09",
        "limit": {
          "context": 128000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "v0-1.5-md",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-06-09",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "v0",
    "npm": "@ai-sdk/vercel"
  },
  "venice": {
    "doc": "https://docs.venice.ai",
    "env": [
      "VENICE_API_KEY"
    ],
    "id": "venice",
    "models": {
      "aion-labs-aion-2-0": {
        "attachment": false,
        "cost": {
          "cache_read": 0.25,
          "input": 1,
          "output": 2
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "aion-labs-aion-2-0",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Aion 2.0",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-24",
        "tool_call": false
      },
      "claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1.2,
          "cache_write": 15,
          "input": 12,
          "output": 60
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-10",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.6,
          "cache_write": 7.5,
          "input": 6,
          "output": 30
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-5",
        "knowledge": "2025-03-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.6,
          "cache_write": 7.5,
          "input": 6,
          "output": 30
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-opus-4-7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.6,
          "cache_write": 7.5,
          "input": 6,
          "output": 30
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-7-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 3.6,
          "cache_write": 45,
          "input": 36,
          "output": 180
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "claude-opus-4-7-fast",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-14",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.6,
          "cache_write": 7.5,
          "input": 6,
          "output": 30
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-28",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-opus-4-8-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 1.2,
          "cache_write": 15,
          "input": 12,
          "output": 60
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "claude-opus-4-8-fast",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-28",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "claude-sonnet-4-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.375,
          "cache_write": 4.69,
          "input": 3.75,
          "output": 18.75
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-5",
        "knowledge": "2025-07-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-4-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.36,
          "cache_write": 4.5,
          "input": 3.6,
          "output": 18
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-4-6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-07-01",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-29",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.16,
          "input": 0.33,
          "output": 0.48
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 160000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-04",
        "structured_output": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.138,
          "output": 0.275
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.33,
          "input": 1.65,
          "output": 3.301
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 0.5,
          "context_over_200k": {
            "cache_read": 0.5,
            "cache_write": 0.5,
            "input": 5,
            "output": 22.5
          },
          "input": 2.5,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.5,
              "cache_write": 0.5,
              "input": 5,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini",
        "id": "gemini-3-1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.155,
          "cache_write": 0.086,
          "input": 1.55,
          "output": 9.45
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini",
        "id": "gemini-3-5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.07,
          "input": 0.7,
          "output": 3.75
        },
        "description": "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs",
        "family": "gemini",
        "id": "gemini-3-flash-preview",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemma-4-uncensored": {
        "attachment": true,
        "cost": {
          "input": 0.1625,
          "output": 0.5
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "gemma-4-uncensored",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 Uncensored",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-13",
        "structured_output": true,
        "tool_call": true
      },
      "google-gemma-3-27b-it": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0.2
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google-gemma-3-27b-it",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 3 27B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-11-04",
        "structured_output": true,
        "tool_call": true
      },
      "google-gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "input": 0.1625,
          "output": 0.5
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google-gemma-4-26b-a4b-it",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 4 26B A4B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google-gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09,
          "input": 0.12,
          "output": 0.36
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google-gemma-4-31b-it",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Google Gemma 4 31B Instruct",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4-20": {
        "attachment": true,
        "cost": {
          "cache_read": 0.23,
          "context_over_200k": {
            "cache_read": 0.45,
            "input": 2.83,
            "output": 5.67
          },
          "input": 1.42,
          "output": 2.83,
          "tiers": [
            {
              "cache_read": 0.45,
              "input": 2.83,
              "output": 5.67,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-20",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 2000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-12",
        "structured_output": true,
        "tool_call": true
      },
      "grok-4-20-multi-agent": {
        "attachment": true,
        "cost": {
          "cache_read": 0.23,
          "context_over_200k": {
            "cache_read": 0.45,
            "input": 2.83,
            "output": 5.67
          },
          "input": 1.42,
          "output": 2.83,
          "tiers": [
            {
              "cache_read": 0.45,
              "input": 2.83,
              "output": 5.67,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4-20-multi-agent",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 2000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Multi-Agent",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-12",
        "structured_output": true,
        "tool_call": false
      },
      "grok-4-3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.23,
          "context_over_200k": {
            "cache_read": 0.45,
            "input": 2.83,
            "output": 5.67
          },
          "input": 1.42,
          "output": 2.83,
          "tiers": [
            {
              "cache_read": 0.45,
              "input": 2.83,
              "output": 5.67,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "grok-4-3",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-build-0-1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 4
          },
          "input": 1,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 4,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "grok-build-0-1",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "hermes-3-llama-3.1-405b": {
        "attachment": false,
        "cost": {
          "input": 1.1,
          "output": 3
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "hermes",
        "id": "hermes-3-llama-3.1-405b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hermes 3 Llama 3.1 405b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-25",
        "tool_call": false
      },
      "kimi-k2-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.22,
          "input": 0.56,
          "output": 3.5
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "kimi-k2-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-04",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "kimi-k2-6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.75,
          "output": 3.5
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "kimi-k2-6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "kimi-k2-7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 0.9,
          "output": 4.3
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "kimi-k2-7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "llama-3.2-3b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.2-3b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-03",
        "tool_call": true
      },
      "llama-3.3-70b": {
        "attachment": false,
        "cost": {
          "input": 0.7,
          "output": 2.8
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "llama-3.3-70b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.3 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-06",
        "tool_call": true
      },
      "mercury-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03125,
          "input": 0.3125,
          "output": 0.9375
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "mercury",
        "id": "mercury-2",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 50000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-20",
        "structured_output": true,
        "tool_call": true
      },
      "minimax-m25": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "input": 0.34,
          "output": 1.19
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax-m25",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m27": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06875,
          "input": 0.375,
          "output": 1.5
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax-m27",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax-m3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal coding model for long-context reasoning and agent tasks",
        "family": "minimax-m3",
        "id": "minimax-m3-preview",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 524288,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M3 Preview",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "tool_call": true
      },
      "mistral-small-2603": {
        "attachment": true,
        "cost": {
          "input": 0.1875,
          "output": 0.75
        },
        "description": "Fast Mistral production model for chat, extraction, and cost-sensitive agents",
        "family": "mistral-small",
        "id": "mistral-small-2603",
        "knowledge": "2025-06",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 4",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "mistral-small-3-2-24b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.09375,
          "output": 0.25
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral-small-3-2-24b-instruct",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small 3.2 24B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-15",
        "structured_output": true,
        "tool_call": true
      },
      "nvidia-nemotron-3-nano-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.075,
          "output": 0.3
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia-nemotron-3-nano-30b-a3b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron 3 Nano 30B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia-nemotron-3-ultra-550b-a55b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1875,
          "input": 0.625,
          "output": 3.125
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia-nemotron-3-ultra-550b-a55b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron 3 Ultra",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-04",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia-nemotron-cascade-2-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.8
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia-nemotron-cascade-2-30b-a3b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron Cascade 2 30B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "olafangensan-glm-4.7-flash-heretic": {
        "attachment": false,
        "cost": {
          "input": 0.14,
          "output": 0.8
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "olafangensan-glm-4.7-flash-heretic",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 200000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash Heretic",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-04",
        "structured_output": true,
        "tool_call": true
      },
      "openai-gpt-4o-2024-11-20": {
        "attachment": true,
        "cost": {
          "input": 3.125,
          "output": 12.5
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "family": "gpt",
        "id": "openai-gpt-4o-2024-11-20",
        "knowledge": "2023-09",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-4o-mini-2024-07-18": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09375,
          "input": 0.1875,
          "output": 0.75
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt",
        "id": "openai-gpt-4o-mini-2024-07-18",
        "knowledge": "2023-09",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai-gpt-52": {
        "attachment": false,
        "cost": {
          "cache_read": 0.219,
          "input": 2.19,
          "output": 17.5
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai-gpt-52",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "input": 272000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-52-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.219,
          "input": 2.19,
          "output": 17.5
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt",
        "id": "openai-gpt-52-codex",
        "knowledge": "2025-08",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "input": 272000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-01-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-53-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.219,
          "input": 2.19,
          "output": 17.5
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "openai-gpt-53-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-54": {
        "attachment": true,
        "cost": {
          "cache_read": 0.313,
          "input": 3.13,
          "output": 18.8
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "openai-gpt-54",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "input": 922000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-54-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09375,
          "input": 0.9375,
          "output": 5.625
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt",
        "id": "openai-gpt-54-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-54-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 75,
            "output": 337.5
          },
          "input": 37.5,
          "output": 225,
          "tiers": [
            {
              "input": 75,
              "output": 337.5,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt",
        "id": "openai-gpt-54-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-55": {
        "attachment": true,
        "cost": {
          "cache_read": 0.625,
          "context_over_200k": {
            "cache_read": 1.25,
            "input": 12.5,
            "output": 56.25
          },
          "input": 6.25,
          "output": 37.5,
          "tiers": [
            {
              "cache_read": 1.25,
              "input": 12.5,
              "output": 56.25,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai-gpt-55",
        "knowledge": "2025-12-01",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "input": 922000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-55-pro": {
        "attachment": true,
        "cost": {
          "input": 37.5,
          "output": 225
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt",
        "id": "openai-gpt-55-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai-gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.3
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai-gpt-oss-120b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenAI GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen-3-6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0625,
          "cache_write": 0.78,
          "context_over_200k": {
            "cache_read": 0.0625,
            "cache_write": 0.78,
            "input": 2.5,
            "output": 7.5
          },
          "input": 0.625,
          "output": 3.75,
          "tiers": [
            {
              "cache_read": 0.0625,
              "cache_write": 0.78,
              "input": 2.5,
              "output": 7.5,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "qwen-3-6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 Plus Uncensored",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen-3-7-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.27,
          "cache_write": 3.35,
          "input": 2.7,
          "output": 8.05
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen-3-7-max",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-22",
        "temperature": true,
        "tool_call": true
      },
      "qwen-3-7-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.15,
            "cache_write": 1.875,
            "input": 1.5,
            "output": 6
          },
          "input": 0.5,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.15,
              "cache_write": 1.875,
              "input": 1.5,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen-3-7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.75
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-235b-a22b-instruct-2507",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "qwen3-235b-a22b-thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.45,
          "output": 3.5
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "qwen3-235b-a22b-thinking-2507",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 235B A22B Thinking 2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "qwen3-5-35b-a3b": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15625,
          "input": 0.3125,
          "output": 1.25
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-5-35b-a3b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 35B A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-5-397b-a17b": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 4.5
        },
        "description": "Large open Qwen multimodal MoE for visual agents and long technical tasks",
        "family": "qwen",
        "id": "qwen3-5-397b-a17b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 397B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-5-9b": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.15
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-5-9b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 9B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.325,
          "output": 3.25
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "qwen3-6-27b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 27B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3-coder-480b-a35b-instruct-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.04,
          "input": 0.35,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "qwen3-coder-480b-a35b-instruct-turbo",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Coder 480B Turbo",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-27",
        "structured_output": true,
        "tool_call": true
      },
      "qwen3-next-80b": {
        "attachment": false,
        "cost": {
          "input": 0.35,
          "output": 1.9
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "qwen3-next-80b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 256000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Next 80b",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-29",
        "structured_output": true,
        "tool_call": true
      },
      "qwen3-vl-235b-a22b": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.5
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "id": "qwen3-vl-235b-a22b",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-01-16",
        "structured_output": true,
        "tool_call": true
      },
      "venice-uncensored-1-2": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.9
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "venice",
        "id": "venice-uncensored-1-2",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Venice Uncensored 1.2",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-01",
        "structured_output": true,
        "tool_call": true
      },
      "venice-uncensored-role-play": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 2
        },
        "description": "Multimodal model for analyzing text, images, documents, and rich media",
        "family": "venice",
        "id": "venice-uncensored-role-play",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Venice Role Play Uncensored",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-20",
        "structured_output": true,
        "tool_call": true
      },
      "xiaomi-mimo-v2-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0625,
          "input": 0.175,
          "output": 0.35
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "xiaomi-mimo-v2-5",
        "knowledge": "2024-12",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai-glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Faster GLM-5 lane for coding agents that need lower latency",
        "family": "glm",
        "id": "z-ai-glm-5-turbo",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 200000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Turbo",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai-glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "input": 1.5,
          "output": 5
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "z-ai-glm-5v-turbo",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 200000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5V Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org-glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "input": 0.43,
          "output": 1.75
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "zai-org-glm-4.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-04-01",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org-glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.55,
          "output": 2.65
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "zai-org-glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org-glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.125,
          "output": 0.5
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm",
        "id": "zai-org-glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-06-11",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org-glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3.2
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "zai-org-glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-11",
        "limit": {
          "context": 198000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org-glm-5-1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.325,
          "input": 1.75,
          "output": 5.5
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org-glm-5-1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-11",
        "limit": {
          "context": 200000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org-glm-5-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai-org-glm-5-2",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Venice AI",
    "npm": "venice-ai-sdk-provider"
  },
  "vercel": {
    "doc": "https://github.com/vercel/ai/tree/5eb85cc45a259553501f535b8ac79a77d0e79223/packages/gateway",
    "env": [
      "AI_GATEWAY_API_KEY"
    ],
    "id": "vercel",
    "models": {
      "alibaba/qwen-3-14b": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.24
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba/qwen-3-14b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 40960,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-14B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen-3-235b": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.88
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba/qwen-3-235b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen-3-30b": {
        "attachment": false,
        "cost": {
          "input": 0.12,
          "output": 0.5
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba/qwen-3-30b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 40960,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-30B-A3B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen-3-32b": {
        "attachment": false,
        "cost": {
          "input": 0.16,
          "output": 0.64
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba/qwen-3-32b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.32B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 38912,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-04-28",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen-3.6-max-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 1.625,
          "input": 1.3,
          "output": 7.8
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "alibaba/qwen-3.6-max-preview",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 240000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 Max Preview",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-235b-a22b-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "alibaba/qwen3-235b-a22b-thinking",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Thinking 2507",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-coder": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 1.5,
          "output": 7.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "alibaba/qwen3-coder",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder 480B A35B Instruct",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-22",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-coder-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "alibaba/qwen3-coder-30b-a3b",
        "knowledge": "2025-04",
        "last_updated": "2025-04",
        "limit": {
          "context": 262144,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Coder 30B A3B Instruct",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-coder-next": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.2
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "alibaba/qwen3-coder-next",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Next",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-22",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 5
        },
        "description": "Hosted Qwen coder for software agents, repo edits, and long-context code",
        "family": "qwen",
        "id": "alibaba/qwen3-coder-plus",
        "knowledge": "2025-04",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Coder Plus",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-embedding-0.6b": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "alibaba/qwen3-embedding-0.6b",
        "last_updated": "2025-11-14",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 0.6B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-14",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/qwen3-embedding-4b": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "alibaba/qwen3-embedding-4b",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 4B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-05",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/qwen3-embedding-8b": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "qwen",
        "id": "alibaba/qwen3-embedding-8b",
        "last_updated": "2025-06-05",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Embedding 8B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-05",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/qwen3-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 6
        },
        "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba/qwen3-max",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-max-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 6
        },
        "description": "Flagship Qwen model for complex reasoning, coding, and agentic workflows",
        "family": "qwen",
        "id": "alibaba/qwen3-max-preview",
        "knowledge": "2025-04",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Max Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-05",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-max-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "alibaba/qwen3-max-thinking",
        "knowledge": "2025-01",
        "last_updated": "2025-01",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3 Max Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-01-23",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-next-80b-a3b-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.2
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "alibaba/qwen3-next-80b-a3b-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-next-80b-a3b-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 1.2
        },
        "description": "Efficient Qwen thinking model for local reasoning, math, and coding agents",
        "family": "qwen",
        "id": "alibaba/qwen3-next-80b-a3b-thinking",
        "knowledge": "2025-09",
        "last_updated": "2025-09",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 Next 80B A3B Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-11",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-vl-235b-a22b-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "alibaba/qwen3-vl-235b-a22b-instruct",
        "last_updated": "2026-05-01",
        "limit": {
          "context": 131072,
          "output": 129024
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL 235B A22B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/qwen3-vl-instruct": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 1.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "alibaba/qwen3-vl-instruct",
        "knowledge": "2025-04",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 131072,
          "output": 129024
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3-vl-thinking": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "alibaba/qwen3-vl-thinking",
        "knowledge": "2025-09",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 VL Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.001,
          "cache_write": 0.125,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "alibaba/qwen3.5-flash",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-24",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.5-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "cache_write": 0.5,
          "input": 0.4,
          "output": 2.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "alibaba/qwen3.5-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-02-16",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 81920,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-16",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.6-27b": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 3.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen3.6",
        "id": "alibaba/qwen3.6-27b",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 27B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-22",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.6-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 0.625,
          "input": 0.5,
          "output": 3
        },
        "description": "Earlier Qwen multimodal workhorse for million-token agent and document tasks",
        "family": "qwen",
        "id": "alibaba/qwen3.6-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.6 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 131072,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-04-02",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.7-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "cache_write": 1.5625,
          "input": 1.25,
          "output": 3.75
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "alibaba/qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 991000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/qwen3.7-plus": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 0.5,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen3.7-plus",
        "id": "alibaba/qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen 3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 262144,
            "min": 1,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "alibaba/wan-v2.5-t2v-preview": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "o",
        "id": "alibaba/wan-v2.5-t2v-preview",
        "last_updated": "2025-09-24",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan v2.5 Text-to-Video Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-24",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/wan-v2.6-i2v": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "o",
        "id": "alibaba/wan-v2.6-i2v",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan v2.6 Image-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/wan-v2.6-i2v-flash": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "o",
        "id": "alibaba/wan-v2.6-i2v-flash",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan v2.6 Image-to-Video Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/wan-v2.6-r2v": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "o",
        "id": "alibaba/wan-v2.6-r2v",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan v2.6 Reference-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/wan-v2.6-r2v-flash": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "o",
        "id": "alibaba/wan-v2.6-r2v-flash",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan v2.6 Reference-to-Video Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "alibaba/wan-v2.6-t2v": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "o",
        "id": "alibaba/wan-v2.6-t2v",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Wan v2.6 Text-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "amazon/nova-2-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "nova",
        "id": "amazon/nova-2-lite",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova 2 Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": false
      },
      "amazon/nova-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.06,
          "output": 0.24
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-lite",
        "id": "amazon/nova-lite",
        "knowledge": "2024-10",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-micro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.00875,
          "input": 0.035,
          "output": 0.14
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "nova-micro",
        "id": "amazon/nova-micro",
        "knowledge": "2024-10",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Micro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "amazon/nova-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 0.8,
          "output": 3.2
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "nova-pro",
        "id": "amazon/nova-pro",
        "knowledge": "2024-10",
        "last_updated": "2024-12-03",
        "limit": {
          "context": 300000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nova Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-03",
        "temperature": true,
        "tool_call": true
      },
      "amazon/titan-embed-text-v2": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "titan-embed",
        "id": "amazon/titan-embed-text-v2",
        "last_updated": "2024-04",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Titan Text Embeddings V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-30",
        "temperature": true,
        "tool_call": false
      },
      "anthropic/claude-3-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.3,
          "input": 0.25,
          "output": 1.25
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "claude-haiku",
        "id": "anthropic/claude-3-haiku",
        "knowledge": "2023-08-31",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 200000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3.5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "family": "claude-haiku",
        "id": "anthropic/claude-3.5-haiku",
        "knowledge": "2024-07-31",
        "last_updated": "2024-10-22",
        "limit": {
          "context": 200000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Haiku",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-04",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "anthropic/claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-07-01",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude lane for lightweight agents, office tasks, and responsive chat",
        "family": "claude-haiku",
        "id": "anthropic/claude-haiku-4.5",
        "interleaved": true,
        "knowledge": "2025-02-28",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.1",
        "knowledge": "2025-03-31",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.5",
        "interleaved": true,
        "knowledge": "2025-03-31",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "High-end Claude for difficult coding, planning, and slower expert reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.6",
        "interleaved": true,
        "knowledge": "2025-05-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Stronger Opus tier for advanced software work and high-stakes reasoning",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-03-31",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.5",
        "knowledge": "2025-07-31",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "context_over_200k": {
            "cache_read": 0.6,
            "cache_write": 7.5,
            "input": 6,
            "output": 22.5
          },
          "input": 3,
          "output": 15,
          "tiers": [
            {
              "cache_read": 0.6,
              "cache_write": 7.5,
              "input": 6,
              "output": 22.5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Claude workhorse for coding agents, careful analysis, and production cost control",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-4.6",
        "interleaved": true,
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          },
          {
            "min": 1024,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 2.5,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-29",
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/trinity-large-preview": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "trinity",
        "id": "arcee-ai/trinity-large-preview",
        "knowledge": "2024-10",
        "last_updated": "2025-01",
        "limit": {
          "context": 131000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Large Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-27",
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/trinity-large-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 0.8999999999999999
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "trinity",
        "id": "arcee-ai/trinity-large-thinking",
        "last_updated": "2026-04-03",
        "limit": {
          "context": 262100,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Large Thinking",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      },
      "arcee-ai/trinity-mini": {
        "attachment": false,
        "cost": {
          "input": 0.045,
          "output": 0.15
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "trinity",
        "id": "arcee-ai/trinity-mini",
        "knowledge": "2024-10",
        "last_updated": "2025-12",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Trinity Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-2-flex": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-2-flex",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 [flex]",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-2-klein-4b": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-2-klein-4b",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 [klein] 4B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-15",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-2-klein-9b": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-2-klein-9b",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 [klein] 9B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-15",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-2-max": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-2-max",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 67300,
          "output": 67300
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 [max]",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-2-pro": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-2-pro",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 67300,
          "output": 67300
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.2 [pro]",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-25",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-kontext-max": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-kontext-max",
        "last_updated": "2025-06",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1 Kontext Max",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-29",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-kontext-pro": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-kontext-pro",
        "last_updated": "2025-06",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1 Kontext Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-29",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-pro-1.0-fill": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-pro-1.0-fill",
        "last_updated": "2024-10",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX.1 Fill [pro]",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-pro-1.1": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-pro-1.1",
        "last_updated": "2024-10",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX1.1 [pro]",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-02",
        "temperature": true,
        "tool_call": false
      },
      "bfl/flux-pro-1.1-ultra": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "bfl/flux-pro-1.1-ultra",
        "last_updated": "2024-11",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "FLUX1.1 [pro] Ultra",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seed-1.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "bytedance/seed-1.6",
        "knowledge": "2024-10",
        "last_updated": "2025-09",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": true
      },
      "bytedance/seed-1.8": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.25,
          "output": 2
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "family": "seed",
        "id": "bytedance/seed-1.8",
        "knowledge": "2024-10",
        "last_updated": "2025-10",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Seed 1.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": true
      },
      "bytedance/seedance-2.0": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "seed",
        "id": "bytedance/seedance-2.0",
        "last_updated": "2026-04-14",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Seedance 2.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-14",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedance-2.0-fast": {
        "attachment": true,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "seed",
        "id": "bytedance/seedance-2.0-fast",
        "last_updated": "2026-04-14",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Seedance 2.0 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-14",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedance-v1.0-pro": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "seed",
        "id": "bytedance/seedance-v1.0-pro",
        "last_updated": "2025-06-11",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Seedance v1.0 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-11",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedance-v1.0-pro-fast": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "seed",
        "id": "bytedance/seedance-v1.0-pro-fast",
        "last_updated": "2025-10-31",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Seedance v1.0 Pro Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-24",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedance-v1.5-pro": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "seed",
        "id": "bytedance/seedance-v1.5-pro",
        "last_updated": "2025-12-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Seedance v1.5 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedream-4.0": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "seed",
        "id": "bytedance/seedream-4.0",
        "last_updated": "2025-08-28",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Seedream 4.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-09",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedream-4.5": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "seed",
        "id": "bytedance/seedream-4.5",
        "last_updated": "2025-11-28",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Seedream 4.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-03",
        "temperature": true,
        "tool_call": false
      },
      "bytedance/seedream-5.0-lite": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "seed",
        "id": "bytedance/seedream-5.0-lite",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Seedream 5.0 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": false
      },
      "cohere/command-a": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
        "family": "command",
        "id": "cohere/command-a",
        "knowledge": "2024-10",
        "last_updated": "2025-03-13",
        "limit": {
          "context": 256000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Command A",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-13",
        "temperature": true,
        "tool_call": true
      },
      "cohere/embed-v4.0": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "cohere-embed",
        "id": "cohere/embed-v4.0",
        "last_updated": "2025-04-15",
        "limit": {
          "context": 128000,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Embed v4.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-15",
        "temperature": true,
        "tool_call": false
      },
      "cohere/rerank-v3.5": {
        "attachment": false,
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "o",
        "id": "cohere/rerank-v3.5",
        "last_updated": "2024-12-02",
        "limit": {
          "context": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Rerank 3.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-02",
        "temperature": true,
        "tool_call": false
      },
      "cohere/rerank-v4-fast": {
        "attachment": false,
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "o",
        "id": "cohere/rerank-v4-fast",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Rerank 4 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "temperature": true,
        "tool_call": false
      },
      "cohere/rerank-v4-pro": {
        "attachment": false,
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "o",
        "id": "cohere/rerank-v4-pro",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Cohere Rerank 4 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-r1": {
        "attachment": false,
        "cost": {
          "input": 1.35,
          "output": 5.4
        },
        "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-r1",
        "knowledge": "2024-07",
        "last_updated": "2025-05-29",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-R1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-01-20",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.27,
          "output": 1.12
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3",
        "knowledge": "2024-07",
        "last_updated": "2024-12-26",
        "limit": {
          "context": 163840,
          "output": 163840
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3 0324",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-26",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1": {
        "attachment": false,
        "cost": {
          "input": 0.6,
          "output": 1.7
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1",
        "knowledge": "2024-07",
        "last_updated": "2025-08-21",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-21",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.1-terminus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.135,
          "input": 0.27,
          "output": 1
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.1-terminus",
        "knowledge": "2025-07",
        "last_updated": "2025-09-22",
        "limit": {
          "context": 131072,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1 Terminus",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-22",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.028,
          "input": 0.28,
          "output": 0.42
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek/deepseek-v3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": false
      },
      "deepseek/deepseek-v3.2-thinking": {
        "attachment": false,
        "cost": {
          "input": 0.62,
          "output": 1.85
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v3.2-thinking",
        "interleaved": true,
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek",
        "id": "deepseek/deepseek-v4-flash",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0036,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek",
        "id": "deepseek/deepseek-v4-pro",
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "input_audio": 1,
          "output": 2.5
        },
        "description": "Fast Gemini workhorse for multimodal apps where latency and price matter",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 0,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-image": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Nano Banana image model for fast generation, edits, and character-consistent assets",
        "family": "gemini-flash",
        "id": "google/gemini-2.5-flash-image",
        "knowledge": "2025-01",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 32768,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana (Gemini 2.5 Flash Image)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-26",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents",
        "family": "gemini-flash-lite",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "max": 24576,
            "min": 512,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "context_over_200k": {
            "cache_read": 0.25,
            "input": 2.5,
            "output": 15
          },
          "input": 1.25,
          "output": 10,
          "tiers": [
            {
              "cache_read": 0.25,
              "input": 2.5,
              "output": 15,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Google's proven reasoning model for coding, math, and multimodal analysis",
        "family": "gemini-pro",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "max": 32768,
            "min": 128,
            "type": "budget_tokens"
          }
        ],
        "release_date": "2025-06-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3-flash",
        "knowledge": "2025-03",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1000000,
          "output": 65000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-pro-image": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-image",
        "knowledge": "2025-03",
        "last_updated": "2025-09",
        "limit": {
          "context": 65536,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Nano Banana Pro (Gemini 3 Pro Image)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts",
        "family": "gemini-pro",
        "id": "google/gemini-3-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2025-11-18",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-image": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 3
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-image",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Gemini 3.1 Flash Image (Nano Banana 2)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-image-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.5,
          "output": 3
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-image-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-26",
        "limit": {
          "context": 131072,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Gemini 3.1 Flash Image Preview (Nano Banana 2)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-26",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1000000,
          "output": 65000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite-image": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-lite-image",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 65536,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-30",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini",
        "id": "google/gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1000000,
          "output": 65000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 12
        },
        "description": "Reasoning-first Gemini preview for agentic coding and complex problem solving",
        "family": "gemini",
        "id": "google/gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-embedding-001": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "gemini-embedding",
        "id": "google/gemini-embedding-001",
        "knowledge": "2025-05",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Embedding 001",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": false
      },
      "google/gemini-embedding-2": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "gemini-embedding",
        "id": "google/gemini-embedding-2",
        "last_updated": "2026-03-23",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini Embedding 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": false
      },
      "google/gemma-4-26b-a4b-it": {
        "attachment": true,
        "cost": {
          "cache_read": 0.015,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
        "family": "gemma",
        "id": "google/gemma-4-26b-a4b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 26B A4B IT",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemma-4-31b-it": {
        "attachment": true,
        "cost": {
          "input": 0.14,
          "output": 0.4
        },
        "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
        "family": "gemma",
        "id": "google/gemma-4-31b-it",
        "last_updated": "2026-04-02",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemma 4 31B IT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-02",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/imagen-4.0-fast-generate-001": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4.0-fast-generate-001",
        "last_updated": "2025-06",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen 4 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-06-01",
        "temperature": true,
        "tool_call": false
      },
      "google/imagen-4.0-generate-001": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4.0-generate-001",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen 4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": false
      },
      "google/imagen-4.0-ultra-generate-001": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "imagen",
        "id": "google/imagen-4.0-ultra-generate-001",
        "last_updated": "2025-05-24",
        "limit": {
          "context": 480,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Imagen 4 Ultra",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-24",
        "temperature": false,
        "tool_call": false
      },
      "google/text-embedding-005": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "google/text-embedding-005",
        "last_updated": "2024-08",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Text Embedding 005",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-01",
        "temperature": true,
        "tool_call": false
      },
      "google/text-multilingual-embedding-002": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "google/text-multilingual-embedding-002",
        "last_updated": "2024-03",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Text Multilingual Embedding 002",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-01",
        "temperature": true,
        "tool_call": false
      },
      "google/veo-3.0-fast-generate-001": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3.0-fast-generate-001",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.0 Fast Generate",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": false
      },
      "google/veo-3.0-generate-001": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3.0-generate-001",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": false
      },
      "google/veo-3.1-fast-generate-001": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3.1-fast-generate-001",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.1 Fast Generate",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": false
      },
      "google/veo-3.1-generate-001": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "veo",
        "id": "google/veo-3.1-generate-001",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Veo 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": false
      },
      "inception/mercury-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.024999999999999998,
          "input": 0.25,
          "output": 0.75
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "family": "mercury",
        "id": "inception/mercury-2",
        "last_updated": "2026-03-06",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury 2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-24",
        "temperature": true,
        "tool_call": true
      },
      "inception/mercury-coder-small": {
        "attachment": false,
        "cost": {
          "input": 0.25,
          "output": 1
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "mercury",
        "id": "inception/mercury-coder-small",
        "last_updated": "2025-02-26",
        "limit": {
          "context": 32000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mercury Coder Small Beta",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-26",
        "temperature": true,
        "tool_call": true
      },
      "interfaze/interfaze-beta": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 3.5
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "interfaze/interfaze-beta",
        "last_updated": "2026-04-29",
        "limit": {
          "context": 1000000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Interfaze Beta",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-07",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v2.5-turbo-i2v": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ling",
        "id": "klingai/kling-v2.5-turbo-i2v",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v2.5 Turbo Image-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v2.5-turbo-t2v": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "ling",
        "id": "klingai/kling-v2.5-turbo-t2v",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v2.5 Turbo Text-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v2.6-i2v": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ling",
        "id": "klingai/kling-v2.6-i2v",
        "last_updated": "2025-12-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v2.6 Image-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-03",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v2.6-motion-control": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "ling",
        "id": "klingai/kling-v2.6-motion-control",
        "last_updated": "2025-12-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v2.6 Motion Control",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-18",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v2.6-t2v": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "ling",
        "id": "klingai/kling-v2.6-t2v",
        "last_updated": "2025-12-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v2.6 Text-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-03",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v3.0-i2v": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "ling",
        "id": "klingai/kling-v3.0-i2v",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v3.0 Image-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v3.0-motion-control": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "ling",
        "id": "klingai/kling-v3.0-motion-control",
        "last_updated": "2026-03-04",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v3.0 Motion Control",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-04",
        "temperature": true,
        "tool_call": false
      },
      "klingai/kling-v3.0-t2v": {
        "attachment": false,
        "description": "Video model for prompt-guided generation, editing, and motion workflows",
        "family": "ling",
        "id": "klingai/kling-v3.0-t2v",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Kling v3.0 Text-to-Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-05",
        "temperature": true,
        "tool_call": false
      },
      "kwaipilot/kat-coder-pro-v1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "kat-coder",
        "id": "kwaipilot/kat-coder-pro-v1",
        "knowledge": "2024-10",
        "last_updated": "2025-10-24",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KAT-Coder-Pro V1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-09",
        "temperature": true,
        "tool_call": false
      },
      "kwaipilot/kat-coder-pro-v2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "kat-coder",
        "id": "kwaipilot/kat-coder-pro-v2",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kat Coder Pro V2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-27",
        "temperature": true,
        "tool_call": true
      },
      "meituan/longcat-flash-chat": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "longcat",
        "id": "meituan/longcat-flash-chat",
        "knowledge": "2024-10",
        "last_updated": "2025-08-30",
        "limit": {
          "context": 128000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LongCat Flash Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-01",
        "temperature": true,
        "tool_call": true
      },
      "meituan/longcat-flash-thinking-2601": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "longcat",
        "id": "meituan/longcat-flash-thinking-2601",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "LongCat Flash Thinking 2601",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-15",
        "temperature": true,
        "tool_call": false
      },
      "meta/llama-3.1-70b": {
        "attachment": false,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.1-70b",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.1-8b": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.22
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.1-8b",
        "knowledge": "2023-12",
        "last_updated": "2024-07-23",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 8B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-23",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-11b": {
        "attachment": true,
        "cost": {
          "input": 0.16,
          "output": 0.16
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta/llama-3.2-11b",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 11B Vision Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.2-1b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.2-1b",
        "knowledge": "2023-12",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 1B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": false
      },
      "meta/llama-3.2-3b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.2-3b",
        "knowledge": "2023-12",
        "last_updated": "2024-09-18",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 3B Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": false
      },
      "meta/llama-3.2-90b": {
        "attachment": true,
        "cost": {
          "input": 0.72,
          "output": 0.72
        },
        "description": "Open Llama multimodal model for image understanding and text reasoning",
        "family": "llama",
        "id": "meta/llama-3.2-90b",
        "knowledge": "2023-12",
        "last_updated": "2024-09-25",
        "limit": {
          "context": 128000,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.2 90B Vision Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-09-25",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-3.3-70b": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta/llama-3.3-70b",
        "knowledge": "2023-12",
        "last_updated": "2024-12-06",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-4-maverick": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for strong reasoning and fast responses",
        "family": "llama",
        "id": "meta/llama-4-maverick",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-4-Maverick-17B-128E-Instruct-FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "meta/llama-4-scout": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta/llama-4-scout",
        "knowledge": "2024-08",
        "last_updated": "2025-04-05",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-4-Scout-17B-16E-Instruct-FP8",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-05",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Efficient open MiniMax model built for coding agents and tool-heavy workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2",
        "interleaved": true,
        "knowledge": "2024-10",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 205000,
          "output": 205000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Earlier MiniMax agent model for practical coding and productivity tasks",
        "family": "minimax",
        "id": "minimax/minimax-m2.1",
        "interleaved": true,
        "knowledge": "2024-10",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1-lightning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2.1-lightning",
        "knowledge": "2024-10",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1 Lightning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Prior MiniMax coding model for agent workflows, office edits, and automation",
        "family": "minimax",
        "id": "minimax/minimax-m2.5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5-highspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "family": "minimax",
        "id": "minimax/minimax-m2.5-highspeed",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5 High Speed",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "minimax/minimax-m2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Minimax M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-highspeed": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.375,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Low-latency M2.7 variant for interactive coding plans and agent loops",
        "family": "minimax",
        "id": "minimax/minimax-m2.7-highspeed",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7 High Speed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax-m3",
        "id": "minimax/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M3",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-31",
        "temperature": true,
        "tool_call": true
      },
      "mistral/codestral": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "Mistral code model for completions, refactors, and developer IDE workflows",
        "family": "codestral",
        "id": "mistral/codestral",
        "knowledge": "2024-10",
        "last_updated": "2025-01-04",
        "limit": {
          "context": 256000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-05-29",
        "temperature": true,
        "tool_call": true
      },
      "mistral/codestral-embed": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "codestral-embed",
        "id": "mistral/codestral-embed",
        "last_updated": "2025-05-28",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Codestral Embed",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-28",
        "temperature": true,
        "tool_call": false
      },
      "mistral/devstral-2": {
        "attachment": false,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral/devstral-2",
        "knowledge": "2024-10",
        "last_updated": "2025-12-09",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-09",
        "temperature": true,
        "tool_call": true
      },
      "mistral/devstral-small": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral/devstral-small",
        "knowledge": "2024-10",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small 1.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-21",
        "temperature": true,
        "tool_call": true
      },
      "mistral/devstral-small-2": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Mistral coding agent model for repository tasks and software engineering workflows",
        "family": "devstral",
        "id": "mistral/devstral-small-2",
        "knowledge": "2024-10",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Devstral Small 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-09",
        "temperature": true,
        "tool_call": true
      },
      "mistral/magistral-medium": {
        "attachment": false,
        "cost": {
          "input": 2,
          "output": 5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral-medium",
        "id": "mistral/magistral-medium",
        "knowledge": "2025-06",
        "last_updated": "2025-03-20",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Medium (latest)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-17",
        "temperature": true,
        "tool_call": true
      },
      "mistral/magistral-small": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Mistral reasoning model for transparent analysis, math, and complex decisions",
        "family": "magistral-small",
        "id": "mistral/magistral-small",
        "knowledge": "2025-06",
        "last_updated": "2025-03-17",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Magistral Small",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-03-17",
        "temperature": true,
        "tool_call": true
      },
      "mistral/ministral-14b": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.2
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral/ministral-14b",
        "knowledge": "2024-10",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 14B",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": false
      },
      "mistral/ministral-3b": {
        "attachment": false,
        "cost": {
          "input": 0.04,
          "output": 0.04
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral/ministral-3b",
        "knowledge": "2024-10",
        "last_updated": "2024-10-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 3B (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/ministral-8b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads",
        "family": "ministral",
        "id": "mistral/ministral-8b",
        "knowledge": "2024-10",
        "last_updated": "2024-10-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ministral 8B (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-10-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-embed": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "mistral-embed",
        "id": "mistral/mistral-embed",
        "last_updated": "2023-12-11",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Embed",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-12-11",
        "temperature": true,
        "tool_call": false
      },
      "mistral/mistral-large-3": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Flagship Mistral model for advanced reasoning, coding, and multilingual work",
        "family": "mistral-large",
        "id": "mistral/mistral-large-3",
        "knowledge": "2024-10",
        "last_updated": "2025-12-02",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Large 3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-02",
        "temperature": true,
        "tool_call": false
      },
      "mistral/mistral-medium": {
        "attachment": true,
        "cost": {
          "input": 0.4,
          "output": 2
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral/mistral-medium",
        "knowledge": "2024-10",
        "last_updated": "2025-05-07",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium 3.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-07",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-medium-3.5": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 7.5
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-medium",
        "id": "mistral/mistral-medium-3.5",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Medium Latest",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-29",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-nemo": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Mistral model for multilingual chat, reasoning, and tool-assisted workflows",
        "family": "mistral-nemo",
        "id": "mistral/mistral-nemo",
        "knowledge": "2024-04",
        "last_updated": "2024-07-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Nemo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "temperature": true,
        "tool_call": true
      },
      "mistral/mistral-small": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
        "family": "mistral-small",
        "id": "mistral/mistral-small",
        "knowledge": "2025-06",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 32000,
          "output": 4000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Mistral Small (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-17",
        "temperature": true,
        "tool_call": true
      },
      "mistral/pixtral-12b": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.15
        },
        "description": "Mistral vision-language model for image understanding and multimodal chat",
        "family": "pixtral",
        "id": "mistral/pixtral-12b",
        "knowledge": "2024-09",
        "last_updated": "2024-09-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral 12B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-09-01",
        "temperature": true,
        "tool_call": true
      },
      "mistral/pixtral-large": {
        "attachment": true,
        "cost": {
          "input": 2,
          "output": 6
        },
        "description": "Mistral's larger vision model for document-heavy image understanding and chat",
        "family": "pixtral",
        "id": "mistral/pixtral-large",
        "knowledge": "2024-11",
        "last_updated": "2024-11-04",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Pixtral Large (latest)",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-11-01",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2": {
        "attachment": false,
        "cost": {
          "input": 0.57,
          "output": 2.3
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2",
        "last_updated": "2025-09-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-11",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.141,
          "input": 0.47,
          "output": 2
        },
        "description": "Thinking Kimi model for slower research passes, planning, and hard technical questions",
        "family": "kimi-thinking",
        "id": "moonshotai/kimi-k2-thinking",
        "interleaved": true,
        "knowledge": "2024-08",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 216144,
          "output": 216144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.6,
          "output": 3
        },
        "description": "Earlier Kimi frontier model for long-context agents, coding, and multimodal work",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.5",
        "interleaved": true,
        "knowledge": "2025-01",
        "last_updated": "2026-01",
        "limit": {
          "context": 262114,
          "output": 262114
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-26",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.6",
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262000,
          "output": 262000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.19,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 256000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code-highspeed": {
        "attachment": true,
        "cost": {
          "cache_read": 0.38,
          "input": 1.9,
          "output": 8
        },
        "description": "Lower-latency Kimi Code variant for interactive edits and coding-agent loops",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code-highspeed",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code High Speed",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "morph/morph-v3-fast": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 1.2
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "morph",
        "id": "morph/morph-v3-fast",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 16000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph v3 Fast",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": false,
        "tool_call": false
      },
      "morph/morph-v3-large": {
        "attachment": false,
        "cost": {
          "input": 0.9,
          "output": 1.9
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "morph",
        "id": "morph/morph-v3-large",
        "last_updated": "2024-08-15",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Morph v3 Large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-15",
        "temperature": false,
        "tool_call": false
      },
      "nvidia/nemotron-3-nano-30b-a3b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.24
        },
        "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-nano-30b-a3b",
        "knowledge": "2024-10",
        "last_updated": "2025-12-15",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Nano 30B A3B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-15",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/nemotron-3-super-120b-a12b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.65
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-super-120b-a12b",
        "last_updated": "2026-03-11",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron 3 Super 120B A12B",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/nemotron-3-ultra-550b-a55b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy",
        "family": "nemotron",
        "id": "nvidia/nemotron-3-ultra-550b-a55b",
        "last_updated": "2026-06-04",
        "limit": {
          "context": 1000000,
          "output": 65000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nemotron 3 Ultra",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-04",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-nano-12b-v2-vl": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 0.6
        },
        "description": "Nemotron multimodal model for visual reasoning and agentic AI workflows",
        "family": "nemotron",
        "id": "nvidia/nemotron-nano-12b-v2-vl",
        "knowledge": "2024-10",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron Nano 12B V2 VL",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-10-28",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/nemotron-nano-9b-v2": {
        "attachment": false,
        "cost": {
          "input": 0.06,
          "output": 0.23
        },
        "description": "Compact Nemotron model for efficient reasoning and deployable AI agents",
        "family": "nemotron",
        "id": "nvidia/nemotron-nano-9b-v2",
        "knowledge": "2024-10",
        "last_updated": "2025-08-18",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Nvidia Nemotron Nano 9B V2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-18",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-3.5-turbo": {
        "attachment": false,
        "cost": {
          "input": 0.5,
          "output": 1.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo",
        "knowledge": "2021-09",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 16385,
          "input": 12289,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-03-01",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-3.5-turbo-instruct": {
        "attachment": false,
        "cost": {
          "input": 1.5,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-3.5-turbo-instruct",
        "knowledge": "2021-09",
        "last_updated": "2023-03-01",
        "limit": {
          "context": 8192,
          "input": 4096,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-3.5 Turbo Instruct",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-09-18",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4-turbo": {
        "attachment": true,
        "cost": {
          "input": 10,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-4-turbo",
        "knowledge": "2023-12",
        "last_updated": "2024-04-09",
        "limit": {
          "context": 128000,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4 Turbo",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Long-lived GPT workhorse for coding, instruction following, and production apps",
        "family": "gpt",
        "id": "openai/gpt-4.1",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.4,
          "output": 1.6
        },
        "description": "Affordable GPT-4.1 lane for fast coding help and structured extraction",
        "family": "gpt-mini",
        "id": "openai/gpt-4.1-mini",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4.1-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks",
        "family": "gpt-nano",
        "id": "openai/gpt-4.1-nano",
        "knowledge": "2024-04",
        "last_updated": "2025-04-14",
        "limit": {
          "context": 1047576,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4.1 nano",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-04-14",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o": {
        "attachment": true,
        "cost": {
          "cache_read": 1.25,
          "input": 2.5,
          "output": 10
        },
        "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
        "family": "gpt",
        "id": "openai/gpt-4o",
        "knowledge": "2023-09",
        "last_updated": "2024-08-06",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.15,
          "output": 0.6
        },
        "description": "Small omni GPT for cheap multimodal assistance and production-scale traffic",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini",
        "knowledge": "2023-09",
        "last_updated": "2024-07-18",
        "limit": {
          "context": 128000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-07-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-4o-mini-search-preview": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "openai/gpt-4o-mini-search-preview",
        "knowledge": "2023-09",
        "last_updated": "2025-01",
        "limit": {
          "context": 128000,
          "input": 111616,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 4o Mini Search Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-12",
        "structured_output": false,
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4o-mini-transcribe": {
        "attachment": false,
        "cost": {
          "input": 1.25,
          "output": 5
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "o-mini",
        "id": "openai/gpt-4o-mini-transcribe",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o mini Transcribe",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-4o-transcribe": {
        "attachment": false,
        "cost": {
          "input": 2.5,
          "output": 10
        },
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "gpt",
        "id": "openai/gpt-4o-transcribe",
        "last_updated": "2024-03-13",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-4o Transcribe",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5",
        "knowledge": "2024-09-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-5-chat",
        "knowledge": "2024-10",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "input": 111616,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "GPT-5 Chat",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-07",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "openai/gpt-5-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-09-15",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-15",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Small GPT-5 for responsive agents, coding help, and everyday automation",
        "family": "gpt-mini",
        "id": "openai/gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.005,
          "input": 0.05,
          "output": 0.4
        },
        "description": "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs",
        "family": "gpt-nano",
        "id": "openai/gpt-5-nano",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5-pro": {
        "attachment": true,
        "cost": {
          "input": 15,
          "output": 120
        },
        "description": "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning",
        "family": "gpt",
        "id": "openai/gpt-5-pro",
        "knowledge": "2024-10",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 400000,
          "input": 128000,
          "output": 272000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high"
            ]
          }
        ],
        "release_date": "2025-10-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Codex GPT for repository edits, code review, and practical software agents",
        "family": "gpt",
        "id": "openai/gpt-5.1-codex",
        "knowledge": "2024-10",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-max": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "openai/gpt-5.1-codex-max",
        "knowledge": "2024-10",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-11-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "openai/gpt-5.1-codex-mini",
        "knowledge": "2024-10",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-instant": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt",
        "id": "openai/gpt-5.1-instant",
        "knowledge": "2024-10",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "input": 111616,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Instant",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-12",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-thinking": {
        "attachment": true,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt",
        "id": "openai/gpt-5.1-thinking",
        "knowledge": "2024-10",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "GPT 5.1 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-12",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Reliable GPT generation for broad coding, writing, and tool-assisted product work",
        "family": "gpt",
        "id": "openai/gpt-5.2",
        "knowledge": "2024-10",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5.2-chat",
        "knowledge": "2024-10",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 128000,
          "input": 111616,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-11",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Code-specialist GPT for repository edits, reviews, and long-running software agents",
        "family": "gpt-codex",
        "id": "openai/gpt-5.2-codex",
        "knowledge": "2024-10",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-18",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows",
        "family": "gpt",
        "id": "openai/gpt-5.2-pro",
        "knowledge": "2024-10",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.2 ",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.3-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "family": "gpt",
        "id": "openai/gpt-5.3-chat",
        "last_updated": "2026-03-06",
        "limit": {
          "context": 128000,
          "input": 111616,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-03",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-02-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Agent-ready GPT for coding and computer-use workflows at a lower cost",
        "family": "gpt",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Strong small GPT for coding subagents, quick tool use, and high-volume work",
        "family": "gpt",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4 Mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation",
        "family": "gpt",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4 Nano",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks",
        "family": "gpt",
        "id": "openai/gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.4 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": false,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1000000,
          "input": 872000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "input": 30,
          "output": 180
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt",
        "id": "openai/gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1000000,
          "input": 872000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT 5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-image-1": {
        "attachment": false,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 40
        },
        "description": "OpenAI image model for production generation, edits, and brand-safe visual workflows",
        "family": "gpt-image",
        "id": "openai/gpt-image-1",
        "last_updated": "2025-04-24",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-25",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-image-1-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 2,
          "output": 8
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai/gpt-image-1-mini",
        "last_updated": "2025-10-06",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 1 Mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-06",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-image-1.5": {
        "attachment": false,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 32
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai/gpt-image-1.5",
        "last_updated": "2025-11-25",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-image-2": {
        "attachment": false,
        "cost": {
          "cache_read": 1.25,
          "input": 5,
          "output": 30
        },
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "gpt-image",
        "id": "openai/gpt-image-2",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "GPT Image 2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-21",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.5
        },
        "description": "Open GPT reasoning model for self-hosted agents and controllable deployments",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "knowledge": "2024-10",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "knowledge": "2024-10",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 131072,
          "input": 122880,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT OSS 20B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-safeguard-20b": {
        "attachment": false,
        "cost": {
          "cache_read": 0.037,
          "input": 0.075,
          "output": 0.3
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-safeguard-20b",
        "knowledge": "2024-10",
        "last_updated": "2024-12-01",
        "limit": {
          "context": 131072,
          "input": 65536,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-safeguard-20b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-10-29",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-realtime-1.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.4,
          "input": 4,
          "output": 16
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "openai/gpt-realtime-1.5",
        "last_updated": "2026-02-23",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "GPT-Realtime-1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-23",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-realtime-2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.4,
          "input": 4,
          "output": 24
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "openai/gpt-realtime-2",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "gpt-realtime-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-07",
        "temperature": true,
        "tool_call": false
      },
      "openai/gpt-realtime-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.6,
          "output": 2.4
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "gpt",
        "id": "openai/gpt-realtime-mini",
        "last_updated": "2025-10-10",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "GPT-Realtime mini",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-10",
        "temperature": true,
        "tool_call": false
      },
      "openai/o1": {
        "attachment": true,
        "cost": {
          "cache_read": 7.5,
          "input": 15,
          "output": 60
        },
        "description": "O-series reasoning model for hard analysis, math, coding, and planning",
        "family": "o",
        "id": "openai/o1",
        "knowledge": "2023-09",
        "last_updated": "2024-12-05",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 2,
          "output": 8
        },
        "description": "Deliberate o-series reasoner for hard math, coding, and multi-step analysis",
        "family": "o",
        "id": "openai/o3",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-deep-research": {
        "attachment": true,
        "cost": {
          "cache_read": 2.5,
          "input": 10,
          "output": 40
        },
        "description": "Research model for long-horizon investigation, synthesis, and analytical reports",
        "family": "o",
        "id": "openai/o3-deep-research",
        "knowledge": "2024-10",
        "last_updated": "2024-06-26",
        "limit": {
          "context": 200000,
          "input": 100000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-deep-research",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium"
            ]
          }
        ],
        "release_date": "2025-06-26",
        "temperature": true,
        "tool_call": true
      },
      "openai/o3-mini": {
        "attachment": false,
        "cost": {
          "cache_read": 0.55,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Smaller o-series reasoner for economical coding, math, and planning tasks",
        "family": "o-mini",
        "id": "openai/o3-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-01-29",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2024-12-20",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/o3-pro": {
        "attachment": true,
        "cost": {
          "input": 20,
          "output": 80
        },
        "description": "High-effort o3 tier for difficult technical reasoning and careful answers",
        "family": "o-pro",
        "id": "openai/o3-pro",
        "knowledge": "2024-10",
        "last_updated": "2025-06-10",
        "limit": {
          "context": 200000,
          "input": 100000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o3 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-10",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/o4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.275,
          "input": 1.1,
          "output": 4.4
        },
        "description": "Fast o-series model for compact reasoning, coding, and tool use",
        "family": "o-mini",
        "id": "openai/o4-mini",
        "knowledge": "2024-05",
        "last_updated": "2025-04-16",
        "limit": {
          "context": 200000,
          "output": 100000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "o4-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-04-16",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/text-embedding-3-large": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "openai/text-embedding-3-large",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8192,
          "input": 6656,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": false
      },
      "openai/text-embedding-3-small": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "openai/text-embedding-3-small",
        "last_updated": "2024-01-25",
        "limit": {
          "context": 8192,
          "input": 6656,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-3-small",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-25",
        "temperature": true,
        "tool_call": false
      },
      "openai/text-embedding-ada-002": {
        "attachment": false,
        "description": "Embedding model for semantic search, retrieval, clustering, and ranking pipelines",
        "family": "text-embedding",
        "id": "openai/text-embedding-ada-002",
        "last_updated": "2022-12-15",
        "limit": {
          "context": 8192,
          "input": 6656,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "text-embedding-ada-002",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2022-12-15",
        "temperature": true,
        "tool_call": false
      },
      "openai/tts-1": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "o",
        "id": "openai/tts-1",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "TTS-1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": false
      },
      "openai/tts-1-hd": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "o",
        "id": "openai/tts-1-hd",
        "last_updated": "2023-11-06",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "TTS-1 HD",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2023-11-06",
        "temperature": true,
        "tool_call": false
      },
      "openai/whisper-1": {
        "attachment": false,
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "whisper",
        "id": "openai/whisper-1",
        "last_updated": "2022-09-21",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Whisper",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2022-09-21",
        "temperature": true,
        "tool_call": false
      },
      "perplexity/sonar": {
        "attachment": true,
        "description": "Sonar search model for current answers, retrieval, and citation-backed chat",
        "family": "sonar",
        "id": "perplexity/sonar",
        "knowledge": "2025-02",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 127000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": true
      },
      "perplexity/sonar-pro": {
        "attachment": true,
        "description": "Advanced Sonar search model for deeper research and cited synthesis",
        "family": "sonar-pro",
        "id": "perplexity/sonar-pro",
        "knowledge": "2025-09",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 200000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": true
      },
      "perplexity/sonar-reasoning-pro": {
        "attachment": false,
        "description": "Web-grounded reasoning model for multi-step research and cited answers",
        "family": "sonar-reasoning",
        "id": "perplexity/sonar-reasoning-pro",
        "knowledge": "2025-09",
        "last_updated": "2025-02-19",
        "limit": {
          "context": 127000,
          "output": 8000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Sonar Reasoning Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "minimal",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-02-19",
        "temperature": true,
        "tool_call": false
      },
      "prodia/flux-fast-schnell": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "flux",
        "id": "prodia/flux-fast-schnell",
        "last_updated": "2026-06-08",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Flux Schnell",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-08-02",
        "temperature": true,
        "tool_call": false
      },
      "quiverai/arrow-1.1": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "o",
        "id": "quiverai/arrow-1.1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Arrow 1.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-16",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v2": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v2",
        "last_updated": "2024-03",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-03-13",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v3": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v3",
        "last_updated": "2024-10",
        "limit": {
          "context": 512,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-10-30",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v4": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v4",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v4-pro": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v4-pro",
        "last_updated": "2026-02-17",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V4 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-17",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v4.1": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v4.1",
        "last_updated": "2026-05-14",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V4.1",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-14",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v4.1-pro": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v4.1-pro",
        "last_updated": "2026-05-14",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V4.1 Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-14",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v4.1-utility": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v4.1-utility",
        "last_updated": "2026-05-14",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V4.1 Utility",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-14",
        "temperature": true,
        "tool_call": false
      },
      "recraft/recraft-v4.1-utility-pro": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "recraft",
        "id": "recraft/recraft-v4.1-utility-pro",
        "last_updated": "2026-05-14",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "image"
          ]
        },
        "name": "Recraft V4.1 Utility Pro",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-14",
        "temperature": true,
        "tool_call": false
      },
      "sakana/fugu-ultra": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Quality-first multi-agent model for hard research, analysis, and competitions",
        "family": "aura",
        "id": "sakana/fugu-ultra",
        "last_updated": "2026-06-21",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Fugu Ultra",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-21",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.09,
          "output": 0.3
        },
        "description": "StepFun flash lane for quick multimodal reasoning and coding assistance",
        "family": "step",
        "id": "stepfun/step-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 262114,
          "output": 262114
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "StepFun 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-29",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.7-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "input": 0.2,
          "output": 1.15
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "family": "step",
        "id": "stepfun/step-3.7-flash",
        "knowledge": "2026-01-01",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": true,
        "tool_call": true
      },
      "voyage/rerank-2.5": {
        "attachment": false,
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "voyage",
        "id": "voyage/rerank-2.5",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voyage Rerank 2.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": false
      },
      "voyage/rerank-2.5-lite": {
        "attachment": false,
        "description": "Reranking model for improving retrieval quality in search and recommendation systems",
        "family": "voyage",
        "id": "voyage/rerank-2.5-lite",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 32000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Voyage Rerank 2.5 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-3-large": {
        "attachment": false,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "voyage",
        "id": "voyage/voyage-3-large",
        "last_updated": "2024-09",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-3-large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-01-07",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-3.5": {
        "attachment": false,
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "voyage",
        "id": "voyage/voyage-3.5",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-3.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-3.5-lite": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "voyage",
        "id": "voyage/voyage-3.5-lite",
        "last_updated": "2025-05-20",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-3.5-lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-05-20",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-4": {
        "attachment": false,
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "voyage",
        "id": "voyage/voyage-4",
        "last_updated": "2026-03-06",
        "limit": {
          "context": 32000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-4",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-15",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-4-large": {
        "attachment": false,
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "voyage",
        "id": "voyage/voyage-4-large",
        "last_updated": "2026-03-06",
        "limit": {
          "context": 32000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-4-large",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-15",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-4-lite": {
        "attachment": false,
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "voyage",
        "id": "voyage/voyage-4-lite",
        "last_updated": "2026-03-06",
        "limit": {
          "context": 32000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-4-lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-15",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-code-2": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "voyage",
        "id": "voyage/voyage-code-2",
        "last_updated": "2024-01",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-code-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-01-01",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-code-3": {
        "attachment": false,
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "family": "voyage",
        "id": "voyage/voyage-code-3",
        "last_updated": "2024-09",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-code-3",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-12-04",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-finance-2": {
        "attachment": false,
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "voyage",
        "id": "voyage/voyage-finance-2",
        "last_updated": "2024-03",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-finance-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-06-03",
        "temperature": true,
        "tool_call": false
      },
      "voyage/voyage-law-2": {
        "attachment": false,
        "description": "General-purpose chat model for instruction following, writing, and analysis",
        "family": "voyage",
        "id": "voyage/voyage-law-2",
        "last_updated": "2024-03",
        "limit": {
          "context": 8192,
          "output": 1536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "voyage-law-2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2024-04-15",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-4.1-fast-non-reasoning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4.1-fast-non-reasoning",
        "knowledge": "2024-10",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-19",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.1-fast-reasoning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "family": "grok",
        "id": "xai/grok-4.1-fast-reasoning",
        "knowledge": "2024-10",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-19",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.20-multi-agent": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4.20-multi-agent",
        "last_updated": "2026-03-23",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Multi-Agent",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.20-multi-agent-beta": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4.20-multi-agent-beta",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Multi Agent Beta",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.20-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4.20-non-reasoning",
        "last_updated": "2026-03-23",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.20-non-reasoning-beta": {
        "attachment": true,
        "cost": {
          "cache_read": 0.4,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4.20-non-reasoning-beta",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Beta Non-Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.20-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4.20-reasoning",
        "last_updated": "2026-03-23",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-10",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.20-reasoning-beta": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "xai/grok-4.20-reasoning-beta",
        "last_updated": "2026-03-13",
        "limit": {
          "context": 2000000,
          "output": 2000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Beta Reasoning",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-11",
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1.25,
          "output": 2.5
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "xai/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-30",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 2
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "xai/grok-build-0.1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-20",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "xai/grok-imagine-image": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "xai/grok-imagine-image",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text",
            "image"
          ]
        },
        "name": "Grok Imagine Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-28",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-imagine-video": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "xai/grok-imagine-video",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Grok Imagine",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-28",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-imagine-video-1.5": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "xai/grok-imagine-video-1.5",
        "last_updated": "2026-06-22",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Grok Imagine Video 1.5",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-06-22",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-imagine-video-1.5-preview": {
        "attachment": false,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "xai/grok-imagine-video-1.5-preview",
        "last_updated": "2026-05-30",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Grok Imagine Video 1.5 Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-05-30",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-stt": {
        "attachment": false,
        "description": "Speech transcription model for accurate audio-to-text and captioning workflows",
        "family": "grok",
        "id": "xai/grok-stt",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok STT",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-tts": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "grok",
        "id": "xai/grok-tts",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "Grok TTS",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-16",
        "temperature": true,
        "tool_call": false
      },
      "xai/grok-voice-think-fast-1.0": {
        "attachment": false,
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "grok",
        "id": "xai/grok-voice-think-fast-1.0",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 0,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "audio"
          ],
          "output": [
            "text",
            "audio"
          ]
        },
        "name": "Grok Voice Think Fast 1.0",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-23",
        "temperature": true,
        "tool_call": false
      },
      "xiaomi/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash",
        "knowledge": "2024-10",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 262144,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 3
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-pro",
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo-v2.5",
        "id": "xiaomi/mimo-v2.5",
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1050000,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo M2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0036,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo-v2.5-pro",
        "id": "xiaomi/mimo-v2.5-pro",
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1050000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "zai/glm-4.5",
        "interleaved": true,
        "knowledge": "2025-07",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 128000,
          "output": 96000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.2,
          "output": 1.1
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "zai/glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 128000,
          "output": 96000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.5v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "zai/glm-4.5v",
        "knowledge": "2025-08",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 66000,
          "output": 16000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "zai/glm-4.6",
        "interleaved": true,
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 96000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.6v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "zai/glm-4.6v",
        "knowledge": "2024-10",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.6v-flash": {
        "attachment": true,
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "zai/glm-4.6v-flash",
        "knowledge": "2024-10",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 128000,
          "output": 24000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V-Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.12,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "zai/glm-4.7",
        "interleaved": true,
        "knowledge": "2024-10",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 200000,
          "output": 120000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "input": 0.07,
          "output": 0.4
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm",
        "id": "zai/glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.06,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "zai/glm-4.7-flashx",
        "interleaved": true,
        "knowledge": "2025-01",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 FlashX",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "input": 0.95,
          "output": 3.15
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "zai/glm-5",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 202800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Faster GLM-5 lane for coding agents that need lower latency",
        "family": "glm",
        "id": "zai/glm-5-turbo",
        "last_updated": "2026-03-16",
        "limit": {
          "context": 202800,
          "output": 131100
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.26,
          "input": 1.3,
          "output": 4.3
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai/glm-5.1",
        "last_updated": "2026-04-07",
        "limit": {
          "context": 202000,
          "output": 202000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "zai/glm-5.2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1040000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5.2-fast": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "input": 3,
          "output": 10.25
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "zai/glm-5.2-fast",
        "last_updated": "2026-06-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-06-16",
        "temperature": true,
        "tool_call": true
      },
      "zai/glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "input": 1.2,
          "output": 4
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "zai/glm-5v-turbo",
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5V Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Vercel AI Gateway",
    "npm": "@ai-sdk/gateway"
  },
  "vivgrid": {
    "api": "https://api.vivgrid.com/v1",
    "doc": "https://docs.vivgrid.com/models",
    "env": [
      "VIVGRID_API_KEY"
    ],
    "id": "vivgrid",
    "models": {
      "deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 0.42
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-v3.2",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "cache_write": 1,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "gemini-3.1-flash-lite-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-03-03",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 4,
            "output": 18
          },
          "input": 2,
          "output": 12,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 4,
              "output": 18,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "family": "gemini-pro",
        "id": "gemini-3.1-pro-preview",
        "knowledge": "2025-01",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "gpt-5-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 2
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5-mini",
        "knowledge": "2024-05-30",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Mini",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.1-codex-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.125,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.1-codex-max",
        "knowledge": "2024-09-30",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Codex Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-13",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.2-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.2-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-01-14",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-14",
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.3-codex": {
        "attachment": false,
        "cost": {
          "cache_read": 0.175,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "family": "gpt-codex",
        "id": "gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-24",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-24",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.25,
          "input": 2.5,
          "output": 15
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "family": "gpt",
        "id": "gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-05",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-05",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.075,
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-mini",
        "id": "gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.4-nano": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "family": "gpt-nano",
        "id": "gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-17",
        "limit": {
          "context": 400000,
          "input": 272000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Nano",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-17",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "provider": {
          "npm": "@ai-sdk/openai-compatible"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Vivgrid",
    "npm": "@ai-sdk/openai"
  },
  "vultr": {
    "api": "https://api.vultrinference.com/v1",
    "doc": "https://api.vultrinference.com/",
    "env": [
      "VULTR_API_KEY"
    ],
    "id": "vultr",
    "models": {
      "MiniMaxAI/MiniMax-M2.7": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.7",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M2.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-21",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/DeepSeek-V3.2-NVFP4": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 1.65
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "nvidia/DeepSeek-V3.2-NVFP4",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/Llama-3.1-Nemotron-Safety-Guard-8B-v3": {
        "attachment": false,
        "cost": {
          "input": 0.01,
          "output": 0.01
        },
        "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
        "family": "llama",
        "id": "nvidia/Llama-3.1-Nemotron-Safety-Guard-8B-v3",
        "knowledge": "2023-12",
        "last_updated": "2025-10-28",
        "limit": {
          "context": 8192,
          "output": 4096
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 Nemotron Safety Guard",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-10-28",
        "temperature": true,
        "tool_call": false
      },
      "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16": {
        "attachment": false,
        "cost": {
          "input": 0.13,
          "output": 0.38
        },
        "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
        "family": "nemotron",
        "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16",
        "knowledge": "2025-05",
        "last_updated": "2026-04-28",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron 3 Nano Omni",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-28",
        "temperature": true,
        "tool_call": true
      },
      "nvidia/Nemotron-Cascade-2-30B-A3B": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
        "family": "nemotron",
        "id": "nvidia/Nemotron-Cascade-2-30B-A3B",
        "knowledge": "2024-07",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron Cascade 2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.85,
          "output": 3.1
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1-FP8",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Vultr",
    "npm": "@ai-sdk/openai-compatible"
  },
  "wafer.ai": {
    "api": "https://pass.wafer.ai/v1",
    "doc": "https://docs.wafer.ai/wafer-pass",
    "env": [
      "WAFER_API_KEY"
    ],
    "id": "wafer.ai",
    "models": {
      "GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 0,
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "GLM-5.1",
        "knowledge": "2025-04",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 202752,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "GLM-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 1.2,
          "output": 4.1
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "GLM-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Kimi-K2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.07,
          "cache_write": 0,
          "input": 0.68,
          "output": 3.15
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "Kimi-K2.6",
        "knowledge": "2025-01",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi-K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen3.5-397B-A17B": {
        "attachment": true,
        "cost": {
          "cache_read": 0.04,
          "cache_write": 0,
          "input": 0.43,
          "output": 2.6
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen3.5-397B-A17B",
        "knowledge": "2025-04",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5-397B-A17B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen3.6-35B-A3B": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0,
          "input": 0.15,
          "output": 1
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "family": "qwen",
        "id": "Qwen3.6-35B-A3B",
        "knowledge": "2025-04",
        "last_updated": "2026-05-30",
        "limit": {
          "context": 256000,
          "input": 229376,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6-35B-A3B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
        "family": "deepseek-flash",
        "id": "deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-05-30",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0,
          "input": 1.74,
          "output": 3.48
        },
        "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
        "family": "deepseek-thinking",
        "id": "deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-05-30",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 0,
          "input": 5,
          "output": 15
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "family": "qwen3.7-max",
        "id": "qwen3.7-max",
        "last_updated": "2026-05-30",
        "limit": {
          "context": 256000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7-Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Wafer",
    "npm": "@ai-sdk/openai-compatible"
  },
  "wandb": {
    "api": "https://api.inference.wandb.ai/v1",
    "doc": "https://docs.wandb.ai/guides/integrations/inference/",
    "env": [
      "WANDB_API_KEY"
    ],
    "id": "wandb",
    "models": {
      "MiniMaxAI/MiniMax-M2.5": {
        "attachment": false,
        "cost": {
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "family": "minimax",
        "id": "MiniMaxAI/MiniMax-M2.5",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 196608,
          "output": 196608
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "OpenPipe/Qwen3-14B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.22
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "OpenPipe/Qwen3-14B-Instruct",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 32768,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "OpenPipe Qwen3 14B Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
        "knowledge": "2025-04",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 235B A22B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-04-28",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-235B-A22B-Thinking-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.1
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "family": "qwen",
        "id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
        "knowledge": "2025-04",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-235B-A22B-Thinking-2507",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-25",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-30B-A3B-Instruct-2507": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "family": "qwen",
        "id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3 30B A3B Instruct 2507",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-29",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "Qwen/Qwen3-Coder-480B-A35B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 1.5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "family": "qwen",
        "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
        "knowledge": "2025-04",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-480B-A35B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek-ai/DeepSeek-V3.1": {
        "attachment": false,
        "cost": {
          "input": 0.55,
          "output": 1.65
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "family": "deepseek",
        "id": "deepseek-ai/DeepSeek-V3.1",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 161000,
          "output": 161000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.1",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-08-21",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.1-70B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.8,
          "output": 0.8
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.1-70B-Instruct",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 3.1 70B",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.1-8B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.22
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.1-8B-Instruct",
        "knowledge": "2023-12",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Meta-Llama-3.1-8B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-07-23",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-3.3-70B-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.71,
          "output": 0.71
        },
        "description": "Open Llama instruction model for multilingual chat, reasoning, and coding",
        "family": "llama",
        "id": "meta-llama/Llama-3.3-70B-Instruct",
        "knowledge": "2023-12",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama-3.3-70B-Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-06",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
        "attachment": false,
        "cost": {
          "input": 0.17,
          "output": 0.66
        },
        "description": "Open multimodal Llama model for long-context analysis and efficient agents",
        "family": "llama",
        "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
        "knowledge": "2024-12",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 64000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Llama 4 Scout 17B 16E Instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2025-01-31",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "microsoft/Phi-4-mini-instruct": {
        "attachment": false,
        "cost": {
          "input": 0.08,
          "output": 0.35
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "family": "phi",
        "id": "microsoft/Phi-4-mini-instruct",
        "knowledge": "2023-10",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 128000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Phi-4-mini-instruct",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2024-12-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/Kimi-K2.5": {
        "attachment": true,
        "cost": {
          "input": 0.5,
          "output": 2.85
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "family": "kimi-k2",
        "id": "moonshotai/Kimi-K2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 0.8
        },
        "description": "Nemotron middle tier for collaborative agents and high-volume reasoning workloads",
        "family": "nemotron",
        "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "NVIDIA Nemotron 3 Super 120B",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-120b": {
        "attachment": false,
        "cost": {
          "input": 0.15,
          "output": 0.6
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-120b",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-120b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-oss-20b": {
        "attachment": false,
        "cost": {
          "input": 0.05,
          "output": 0.2
        },
        "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
        "family": "gpt-oss",
        "id": "openai/gpt-oss-20b",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 131072,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "gpt-oss-20b",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5-FP8": {
        "attachment": false,
        "cost": {
          "input": 1,
          "output": 3.2
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "zai-org/GLM-5-FP8",
        "last_updated": "2026-03-12",
        "limit": {
          "context": 200000,
          "output": 200000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-02-11",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "zai-org/GLM-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "zai-org/GLM-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Weights & Biases",
    "npm": "@ai-sdk/openai-compatible"
  },
  "xai": {
    "doc": "https://docs.x.ai/docs/models",
    "env": [
      "XAI_API_KEY"
    ],
    "id": "xai",
    "models": {
      "grok-4.20-0309-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4.20-0309-non-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Non-Reasoning)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4.20-0309-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Reasoning Grok for document-heavy analysis and long-horizon tool use",
        "family": "grok",
        "id": "grok-4.20-0309-reasoning",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 (Reasoning)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-4.20-multi-agent-0309": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "family": "grok",
        "id": "grok-4.20-multi-agent-0309",
        "last_updated": "2026-03-09",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.20 Multi-Agent",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-09",
        "structured_output": true,
        "temperature": true,
        "tool_call": false
      },
      "grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 4
          },
          "input": 1,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 4,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "grok-build-0.1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "grok-imagine-image": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "grok-imagine-image",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 8000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "image",
            "pdf"
          ]
        },
        "name": "Grok Imagine Image",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-28",
        "temperature": false,
        "tool_call": false
      },
      "grok-imagine-image-quality": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "grok-imagine-image-quality",
        "last_updated": "2026-04-03",
        "limit": {
          "context": 8000,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "image",
            "pdf"
          ]
        },
        "name": "Grok Imagine Image Quality",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-04-03",
        "temperature": false,
        "tool_call": false
      },
      "grok-imagine-video": {
        "attachment": true,
        "description": "Image model for prompt-driven generation, editing, and visual design workflows",
        "family": "grok",
        "id": "grok-imagine-video",
        "last_updated": "2026-01-28",
        "limit": {
          "context": 1024,
          "output": 0
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "video"
          ]
        },
        "name": "Grok Imagine Video",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-01-28",
        "temperature": false,
        "tool_call": false
      }
    },
    "name": "xAI",
    "npm": "@ai-sdk/xai"
  },
  "xiaomi": {
    "api": "https://api.xiaomimimo.com/v1",
    "doc": "https://platform.xiaomimimo.com/#/docs",
    "env": [
      "XIAOMI_API_KEY"
    ],
    "id": "xiaomi",
    "models": {
      "mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo",
        "id": "mimo-v2-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12-01",
        "last_updated": "2026-06-24",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-16",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Legacy model retained for compatibility with older integrations",
        "family": "mimo",
        "id": "mimo-v2-omni",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-06-24",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0036,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-06-24",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-06-24",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0036,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-06-24",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro-ultraspeed": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0108,
          "input": 1.305,
          "output": 2.61
        },
        "description": "MiMo pro model for strong multimodal reasoning and agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro-ultraspeed",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro-UltraSpeed",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-08",
        "status": "beta",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Xiaomi",
    "npm": "@ai-sdk/openai-compatible"
  },
  "xiaomi-token-plan-ams": {
    "api": "https://token-plan-ams.xiaomimimo.com/v1",
    "doc": "https://platform.xiaomimimo.com/#/docs",
    "env": [
      "XIAOMI_API_KEY"
    ],
    "id": "xiaomi-token-plan-ams",
    "models": {
      "mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo",
        "id": "mimo-v2-omni",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-tts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2-tts",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2-TTS",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-18",
        "tool_call": false
      },
      "mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-tts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      },
      "mimo-v2.5-tts-voiceclone": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts-voiceclone",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS-VoiceClone",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      },
      "mimo-v2.5-tts-voicedesign": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts-voicedesign",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS-VoiceDesign",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      }
    },
    "name": "Xiaomi Token Plan (Europe)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "xiaomi-token-plan-cn": {
    "api": "https://token-plan-cn.xiaomimimo.com/v1",
    "doc": "https://platform.xiaomimimo.com/#/docs",
    "env": [
      "XIAOMI_API_KEY"
    ],
    "id": "xiaomi-token-plan-cn",
    "models": {
      "mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo",
        "id": "mimo-v2-omni",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-tts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2-tts",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2-TTS",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-18",
        "tool_call": false
      },
      "mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-tts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      },
      "mimo-v2.5-tts-voiceclone": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts-voiceclone",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS-VoiceClone",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      },
      "mimo-v2.5-tts-voicedesign": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts-voicedesign",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS-VoiceDesign",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      }
    },
    "name": "Xiaomi Token Plan (China)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "xiaomi-token-plan-sgp": {
    "api": "https://token-plan-sgp.xiaomimimo.com/v1",
    "doc": "https://platform.xiaomimimo.com/#/docs",
    "env": [
      "XIAOMI_API_KEY"
    ],
    "id": "xiaomi-token-plan-sgp",
    "models": {
      "mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo",
        "id": "mimo-v2-omni",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 262144,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "mimo-v2-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "status": "deprecated",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2-tts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2-tts",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2-TTS",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-03-18",
        "tool_call": false
      },
      "mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "mimo-v2.5-tts": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      },
      "mimo-v2.5-tts-voiceclone": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts-voiceclone",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS-VoiceClone",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      },
      "mimo-v2.5-tts-voicedesign": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Speech generation model for controllable voice, narration, and audio delivery",
        "family": "mimo",
        "id": "mimo-v2.5-tts-voicedesign",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 8192,
          "output": 8192
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "audio"
          ]
        },
        "name": "MiMo-V2.5-TTS-VoiceDesign",
        "open_weights": true,
        "reasoning": false,
        "release_date": "2026-04-22",
        "tool_call": false
      }
    },
    "name": "Xiaomi Token Plan (Singapore)",
    "npm": "@ai-sdk/openai-compatible"
  },
  "xpersona": {
    "api": "https://www.xpersona.co/v1",
    "doc": "https://www.xpersona.co/docs",
    "env": [
      "XPERSONA_API_KEY"
    ],
    "id": "xpersona",
    "models": {
      "xpersona-frieren-coder": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 6,
          "reasoning": 6
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "xpersona-frieren-coder",
        "knowledge": "2025-12-30",
        "last_updated": "2026-05-25",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Xpersona Frieren 1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-01",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "xpersona-gpt-5.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.3,
          "input": 3,
          "output": 18,
          "reasoning": 18
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "family": "gpt",
        "id": "xpersona-gpt-5.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-12-30",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high",
              "xhigh",
              "max"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      }
    },
    "name": "Xpersona",
    "npm": "@ai-sdk/openai-compatible"
  },
  "zai": {
    "api": "https://api.z.ai/api/paas/v4",
    "doc": "https://docs.z.ai/guides/overview/pricing",
    "env": [
      "ZHIPU_API_KEY"
    ],
    "id": "zai",
    "models": {
      "glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0,
          "input": 0.2,
          "output": 1.1
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "glm-4.5-flash",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5v": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.5v",
        "knowledge": "2025-04",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 64000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6v": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm-flash",
        "id": "glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0,
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "glm-4.7-flashx",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-FlashX",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 1,
          "output": 3.2
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-12",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.24,
          "cache_write": 0,
          "input": 1.2,
          "output": 4
        },
        "description": "Faster GLM-5 lane for coding agents that need lower latency",
        "family": "glm",
        "id": "glm-5-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-16",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-07",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.24,
          "cache_write": 0,
          "input": 1.2,
          "output": 4
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Z.AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "zai-coding-plan": {
    "api": "https://api.z.ai/api/coding/paas/v4",
    "doc": "https://docs.z.ai/devpack/overview",
    "env": [
      "ZHIPU_API_KEY"
    ],
    "id": "zai-coding-plan",
    "models": {
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-5-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-16",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Z.AI Coding Plan",
    "npm": "@ai-sdk/openai-compatible"
  },
  "zeldoc": {
    "api": "https://api.zeldoc.ai/v1",
    "doc": "https://docs.zeldoc.ai",
    "env": [
      "ZELDOC_API_KEY"
    ],
    "id": "zeldoc",
    "models": {
      "z-code": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "z-code",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01",
        "last_updated": "2026-04-15",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Z-Code",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-04-15",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Zeldoc",
    "npm": "@ai-sdk/openai-compatible"
  },
  "zenmux": {
    "api": "https://zenmux.ai/api/v1",
    "doc": "https://docs.zenmux.ai",
    "env": [
      "ZENMUX_API_KEY"
    ],
    "id": "zenmux",
    "models": {
      "anthropic/claude-3.5-haiku": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 1,
          "input": 0.8,
          "output": 4
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "anthropic/claude-3.5-haiku",
        "knowledge": "2025-01-01",
        "last_updated": "2024-11-04",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.5 Haiku",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": false,
        "release_date": "2024-11-04",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-3.7-sonnet": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-3.7-sonnet",
        "knowledge": "2025-01-01",
        "last_updated": "2025-02-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude 3.7 Sonnet",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-02-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-fable-5": {
        "attachment": true,
        "cost": {
          "cache_read": 1,
          "cache_write": 12.5,
          "input": 10,
          "output": 50
        },
        "description": "Claude model for creative writing, analysis, and controlled agent workflows",
        "family": "claude-fable",
        "id": "anthropic/claude-fable-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-09",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Fable 5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-09",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-haiku-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
        "id": "anthropic/claude-haiku-4.5",
        "knowledge": "2025-01-01",
        "last_updated": "2025-10-15",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Haiku 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": false,
        "release_date": "2025-10-15",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4",
        "knowledge": "2025-01-01",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 200000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.1": {
        "attachment": true,
        "cost": {
          "cache_read": 1.5,
          "cache_write": 18.75,
          "input": 15,
          "output": 75
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.1",
        "knowledge": "2025-01-01",
        "last_updated": "2025-08-05",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.1",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-05",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.5",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-24",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-24",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.6",
        "knowledge": "2025-05-31",
        "last_updated": "2026-02-06",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.6",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-06",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-opus-4.7": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
        "id": "anthropic/claude-opus-4.7",
        "knowledge": "2026-01-31",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.7",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-16",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-opus-4.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 6.25,
          "input": 5,
          "output": 25
        },
        "description": "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents",
        "family": "claude-opus",
        "id": "anthropic/claude-opus-4.8",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Opus 4.8",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-28",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4",
        "knowledge": "2025-01-01",
        "last_updated": "2025-05-22",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-05-22",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4.5",
        "knowledge": "2025-01-01",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-4.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.3,
          "cache_write": 3.75,
          "input": 3,
          "output": 15
        },
        "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
        "id": "anthropic/claude-sonnet-4.6",
        "knowledge": "2025-08-31",
        "last_updated": "2026-02-18",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 4.6",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-18",
        "temperature": true,
        "tool_call": true
      },
      "anthropic/claude-sonnet-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 4,
          "input": 2,
          "output": 10
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-5",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      },
      "anthropic/claude-sonnet-5-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Everyday Claude agent model for coding, planning, browsing, and general work",
        "family": "claude-sonnet",
        "id": "anthropic/claude-sonnet-5-free",
        "knowledge": "2026-01-31",
        "last_updated": "2026-06-30",
        "limit": {
          "context": 1000000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Claude Sonnet 5 (Free)",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-06-30",
        "temperature": false,
        "tool_call": true
      },
      "baidu/ernie-5.0-thinking-preview": {
        "attachment": true,
        "cost": {
          "input": 0.84,
          "output": 3.37
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "baidu/ernie-5.0-thinking-preview",
        "knowledge": "2025-01-01",
        "last_updated": "2026-01-22",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "ERNIE 5.0",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-22",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-chat": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "input": 0.28,
          "output": 0.42
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-chat",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-01",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2 (Non-thinking Mode)",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-12-01",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2": {
        "attachment": false,
        "cost": {
          "input": 0.28,
          "output": 0.43
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-05",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V3.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-05",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v3.2-exp": {
        "attachment": false,
        "cost": {
          "input": 0.22,
          "output": 0.33
        },
        "description": "DeepSeek chat model for instruction following, coding, and analysis",
        "id": "deepseek/deepseek-v3.2-exp",
        "knowledge": "2025-01-01",
        "last_updated": "2025-09-29",
        "limit": {
          "context": 163000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek-V3.2-Exp",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-29",
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.0028,
          "input": 0.14,
          "output": 0.28
        },
        "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
        "family": "deepseek-flash",
        "id": "deepseek/deepseek-v4-flash",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "deepseek/deepseek-v4-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.003625,
          "input": 0.435,
          "output": 0.87
        },
        "description": "Open MoE flagship with million-token context for coding and long agent runs",
        "family": "deepseek-thinking",
        "id": "deepseek/deepseek-v4-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-05",
        "last_updated": "2026-04-24",
        "limit": {
          "context": 1000000,
          "output": 384000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "DeepSeek V4 Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          },
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-24",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.07,
          "cache_write": 1,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-2.5-flash",
        "knowledge": "2025-01-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 1,
          "input": 0.1,
          "output": 0.4
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-2.5-flash-lite",
        "knowledge": "2025-01-01",
        "last_updated": "2025-07-22",
        "limit": {
          "context": 1048000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Flash Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-22",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-2.5-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.31,
          "cache_write": 4.5,
          "input": 1.25,
          "output": 10
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-2.5-pro",
        "knowledge": "2025-01-01",
        "last_updated": "2025-06-17",
        "limit": {
          "context": 1048000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 2.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-06-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3-flash-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 1,
          "input": 0.5,
          "output": 3
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "id": "google/gemini-3-flash-preview",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-17",
        "limit": {
          "context": 1048000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf",
            "audio"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3 Flash Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-17",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.025,
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "family": "gemini-flash-lite",
        "id": "google/gemini-3.1-flash-lite",
        "knowledge": "2025-01",
        "last_updated": "2026-05-07",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-07",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-flash-lite-preview": {
        "attachment": true,
        "cost": {
          "input": 0.25,
          "output": 1.5
        },
        "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
        "id": "google/gemini-3.1-flash-lite-preview",
        "last_updated": "2025-03-20",
        "limit": {
          "context": 1050000,
          "output": 65530
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Flash Lite Preview",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-03-20",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.1-pro-preview": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 4.5,
          "input": 2,
          "output": 12
        },
        "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
        "id": "google/gemini-3.1-pro-preview",
        "knowledge": "2026-02-19",
        "last_updated": "2026-02-19",
        "limit": {
          "context": 1048000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.1 Pro Preview",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-19",
        "temperature": true,
        "tool_call": true
      },
      "google/gemini-3.5-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.15,
          "input": 1.5,
          "output": 9
        },
        "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
        "family": "gemini-flash",
        "id": "google/gemini-3.5-flash",
        "knowledge": "2025-01",
        "last_updated": "2026-05-19",
        "limit": {
          "context": 1048576,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "audio",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Gemini 3.5 Flash",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-19",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ling-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.56,
          "output": 2.24
        },
        "description": "Tool-capable chat model for instruction following and agentic application workflows",
        "id": "inclusionai/ling-1t",
        "knowledge": "2025-01-01",
        "last_updated": "2025-10-09",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ling-1T",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-10-09",
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ring-1t": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "input": 0.56,
          "output": 2.24
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "inclusionai/ring-1t",
        "knowledge": "2025-01-01",
        "last_updated": "2025-10-12",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Ring-1T",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-12",
        "temperature": true,
        "tool_call": true
      },
      "inclusionai/ring-2.6-1t": {
        "attachment": true,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 2.5
        },
        "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
        "id": "inclusionai/ring-2.6-1t",
        "knowledge": "2025-12-31",
        "last_updated": "2026-05-14",
        "limit": {
          "context": 262000,
          "output": 65000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "inclusionAI: Ring-2.6-1T",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-05-07",
        "temperature": true,
        "tool_call": true
      },
      "kuaishou/kat-coder-pro-v2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.3,
          "output": 1.2
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "kuaishou/kat-coder-pro-v2",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 256000,
          "output": 80000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "KAT-Coder-Pro-V2",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-30",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.38,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2",
        "knowledge": "2025-01-01",
        "last_updated": "2025-10-27",
        "limit": {
          "context": 204000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-10-27",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.38,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.1",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.1",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0.375,
          "input": 0.3,
          "output": 1.2
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.5",
        "knowledge": "2025-01-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.5-lightning": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "cache_write": 0.75,
          "input": 0.6,
          "output": 4.8
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "id": "minimax/minimax-m2.5-lightning",
        "knowledge": "2025-01-01",
        "last_updated": "2026-02-13",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.5 highspeed",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-13",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7": {
        "attachment": true,
        "cost": {
          "input": 0.3055,
          "output": 1.2219
        },
        "description": "MiniMax model for chat, coding, office work, and agentic tasks",
        "id": "minimax/minimax-m2.7",
        "knowledge": "2025-01-01",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 204800,
          "output": 131070
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m2.7-highspeed": {
        "attachment": true,
        "cost": {
          "input": 0.611,
          "output": 2.4439
        },
        "description": "High-speed MiniMax model for low-latency coding and agent workflows",
        "id": "minimax/minimax-m2.7-highspeed",
        "knowledge": "2025-01-01",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 204800,
          "output": 131070
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax M2.7 highspeed",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "minimax/minimax-m3": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 2.4
        },
        "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
        "family": "minimax",
        "id": "minimax/minimax-m3",
        "last_updated": "2026-06-01",
        "limit": {
          "context": 512000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiniMax-M3",
        "open_weights": true,
        "provider": {
          "api": "https://zenmux.ai/api/anthropic/v1",
          "npm": "@ai-sdk/anthropic"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-01",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-0905": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi model for long-context chat, coding, and agentic reasoning",
        "id": "moonshotai/kimi-k2-0905",
        "knowledge": "2025-01-01",
        "last_updated": "2025-09-04",
        "limit": {
          "context": 262000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 0905",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-09-04",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 0.6,
          "output": 2.5
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "moonshotai/kimi-k2-thinking",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2-thinking-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0.15,
          "input": 1.15,
          "output": 8
        },
        "description": "Kimi reasoning model for long-horizon research, planning, and tool use",
        "id": "moonshotai/kimi-k2-thinking-turbo",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-06",
        "limit": {
          "context": 262000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2 Thinking Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-11-06",
        "temperature": true,
        "tool_call": true
      },
      "moonshotai/kimi-k2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1,
          "input": 0.58,
          "output": 3.02
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "moonshotai/kimi-k2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2026-01-27",
        "limit": {
          "context": 262000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-27",
        "temperature": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.6": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
        "id": "moonshotai/kimi-k2.6",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 262140,
          "output": 262140
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-20",
        "temperature": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.16,
          "input": 0.95,
          "output": 4
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "moonshotai/kimi-k2.7-code-free": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
        "family": "kimi-k2",
        "id": "moonshotai/kimi-k2.7-code-free",
        "knowledge": "2025-01",
        "last_updated": "2026-06-12",
        "limit": {
          "context": 262144,
          "output": 262144
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Kimi K2.7 Code (Free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-06-12",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5",
        "knowledge": "2025-01-01",
        "last_updated": "2025-08-07",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-08-07",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5-codex",
        "knowledge": "2025-01-01",
        "last_updated": "2025-09-23",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5 Codex",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-09-23",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 1.25,
          "output": 10
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5.1",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-chat": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 1.25,
          "output": 10
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-5.1-chat",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "pdf",
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1 Chat",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": false,
        "release_date": "2025-11-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.12,
          "input": 1.25,
          "output": 10
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.1-codex",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.1-codex-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.25,
          "output": 2
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.1-codex-mini",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-13",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.1-Codex-Mini",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-13",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.2": {
        "attachment": true,
        "cost": {
          "cache_read": 0.17,
          "input": 1.75,
          "output": 14
        },
        "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
        "id": "openai/gpt-5.2",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-codex": {
        "attachment": true,
        "cost": {
          "cache_read": 0.17,
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.2-codex",
        "knowledge": "2025-01-01",
        "last_updated": "2026-01-15",
        "limit": {
          "context": 400000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Codex",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-01-15",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.2-pro": {
        "attachment": true,
        "cost": {
          "input": 21,
          "output": 168
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.2-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2025-12-11",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.2-Pro",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2025-12-11",
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.3-chat": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Chat-tuned GPT model for conversational assistance, writing, and tool workflows",
        "id": "openai/gpt-5.3-chat",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 128000,
          "output": 16380
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Chat",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.3-codex": {
        "attachment": true,
        "cost": {
          "input": 1.75,
          "output": 14
        },
        "description": "Coding-optimized GPT model for repository edits, reviews, and agentic software work",
        "id": "openai/gpt-5.3-codex",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.3 Codex",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4": {
        "attachment": true,
        "cost": {
          "input": 3.75,
          "output": 18.75
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.4",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4-mini": {
        "attachment": true,
        "cost": {
          "input": 0.75,
          "output": 4.5
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-mini",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Mini",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4-nano": {
        "attachment": false,
        "cost": {
          "input": 0.2,
          "output": 1.25
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.4-nano",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Nano",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.4-pro": {
        "attachment": true,
        "cost": {
          "input": 45,
          "output": 225
        },
        "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
        "id": "openai/gpt-5.4-pro",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 1050000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.4 Pro",
        "open_weights": false,
        "provider": {
          "api": "https://zenmux.ai/api/v1",
          "npm": "@ai-sdk/openai"
        },
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "context_over_200k": {
            "cache_read": 1,
            "input": 10,
            "output": 45
          },
          "input": 5,
          "output": 30,
          "tiers": [
            {
              "cache_read": 1,
              "input": 10,
              "output": 45,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Default frontier GPT for coding, computer use, research, and knowledge work",
        "experimental": {
          "modes": {
            "fast": {
              "cost": {
                "cache_read": 1.25,
                "input": 12.5,
                "output": 75
              },
              "provider": {
                "body": {
                  "service_tier": "priority"
                }
              }
            }
          }
        },
        "family": "gpt",
        "id": "openai/gpt-5.5",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "openai/gpt-5.5-instant": {
        "attachment": true,
        "cost": {
          "cache_read": 0.5,
          "input": 5,
          "output": 30
        },
        "description": "Compact GPT model for low-latency assistance and high-volume workloads",
        "id": "openai/gpt-5.5-instant",
        "knowledge": "2025-12-01",
        "last_updated": "2026-05-28",
        "limit": {
          "context": 400000,
          "input": 400000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Instant",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-05",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "openai/gpt-5.5-pro": {
        "attachment": true,
        "cost": {
          "context_over_200k": {
            "input": 60,
            "output": 270
          },
          "input": 30,
          "output": 180,
          "tiers": [
            {
              "input": 60,
              "output": 270,
              "tier": {
                "size": 272000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding",
        "family": "gpt-pro",
        "id": "openai/gpt-5.5-pro",
        "knowledge": "2025-12-01",
        "last_updated": "2026-04-23",
        "limit": {
          "context": 1050000,
          "input": 922000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GPT-5.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "medium",
              "high",
              "xhigh"
            ]
          }
        ],
        "release_date": "2026-04-23",
        "structured_output": true,
        "temperature": false,
        "tool_call": true
      },
      "qwen/qwen3-coder-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1,
          "cache_write": 1.25,
          "input": 1,
          "output": 5
        },
        "description": "Qwen coding model for software agents, repository edits, and code reasoning",
        "id": "qwen/qwen3-coder-plus",
        "knowledge": "2025-01-01",
        "last_updated": "2025-07-23",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Coder-Plus",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-07-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3-max": {
        "attachment": false,
        "cost": {
          "input": 1.2,
          "output": 6
        },
        "description": "Qwen reasoning model for deliberate problem solving, math, and coding",
        "id": "qwen/qwen3-max",
        "knowledge": "2025-01-01",
        "last_updated": "2026-01-23",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3-Max-Thinking",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-23",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-flash": {
        "attachment": true,
        "cost": {
          "input": 0.1,
          "output": 0.4
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-flash",
        "knowledge": "2025-01-01",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 1020000,
          "output": 1020000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.5-plus": {
        "attachment": true,
        "cost": {
          "input": 0.8,
          "output": 4.8
        },
        "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
        "id": "qwen/qwen3.5-plus",
        "knowledge": "2025-01-01",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.5 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.6-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.05,
          "cache_write": 0.625,
          "context_over_200k": {
            "cache_read": 0.2,
            "cache_write": 2.5,
            "input": 2,
            "output": 6
          },
          "input": 0.5,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.2,
              "cache_write": 2.5,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
        "id": "qwen/qwen3.6-plus",
        "last_updated": "2026-03-30",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.6-Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-30",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.7-max": {
        "attachment": false,
        "cost": {
          "cache_read": 0.5,
          "cache_write": 3.125,
          "input": 2.5,
          "output": 7.5
        },
        "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
        "family": "qwen",
        "id": "qwen/qwen3.7-max",
        "last_updated": "2026-05-21",
        "limit": {
          "context": 1000000,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Max",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-05-21",
        "temperature": true,
        "tool_call": true
      },
      "qwen/qwen3.7-plus": {
        "attachment": false,
        "cost": {
          "cache_read": 0.08,
          "cache_write": 0.5,
          "context_over_200k": {
            "cache_read": 0.24,
            "cache_write": 1.5,
            "input": 1.2,
            "output": 4.8
          },
          "input": 0.4,
          "output": 1.6,
          "tiers": [
            {
              "cache_read": 0.24,
              "cache_write": 1.5,
              "input": 1.2,
              "output": 4.8,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding",
        "family": "qwen",
        "id": "qwen/qwen3.7-plus",
        "knowledge": "2025-04",
        "last_updated": "2026-06-02",
        "limit": {
          "context": 1000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Qwen3.7 Plus",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-06-02",
        "temperature": true,
        "tool_call": true
      },
      "sapiens-ai/agnes-1.5-lite": {
        "attachment": true,
        "cost": {
          "input": 0.12,
          "output": 0.6
        },
        "description": "Efficient model for low-latency assistance, extraction, and routine automation",
        "id": "sapiens-ai/agnes-1.5-lite",
        "last_updated": "2026-03-26",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agnes 1.5 Lite",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-26",
        "temperature": true,
        "tool_call": true
      },
      "sapiens-ai/agnes-1.5-pro": {
        "attachment": false,
        "cost": {
          "input": 0.16,
          "output": 0.8
        },
        "description": "Flagship model for demanding analysis, coding, and production agent workflows",
        "id": "sapiens-ai/agnes-1.5-pro",
        "last_updated": "2026-03-21",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Agnes 1.5 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-21",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3": {
        "attachment": true,
        "cost": {
          "input": 0.21,
          "output": 0.57
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun/step-3",
        "knowledge": "2025-01-01",
        "last_updated": "2025-07-31",
        "limit": {
          "context": 65536,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step-3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-31",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.5-flash": {
        "attachment": false,
        "cost": {
          "input": 0.1,
          "output": 0.3
        },
        "description": "StepFun flash model for efficient multimodal reasoning, coding, and tool use",
        "id": "stepfun/step-3.5-flash",
        "knowledge": "2025-01-01",
        "last_updated": "2026-02-02",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.5 Flash",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-02-02",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.7-flash": {
        "attachment": true,
        "cost": {
          "input": 0.2,
          "output": 1.15
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "id": "stepfun/step-3.7-flash",
        "knowledge": "2026-01-01",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": true
      },
      "stepfun/step-3.7-flash-free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
        "id": "stepfun/step-3.7-flash-free",
        "knowledge": "2026-01-01",
        "last_updated": "2026-05-29",
        "limit": {
          "context": 256000,
          "input": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Step 3.7 Flash (Free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-05-29",
        "temperature": true,
        "tool_call": true
      },
      "tencent/hy3-preview": {
        "attachment": false,
        "cost": {
          "cache_read": 0.058,
          "cache_write": 0,
          "input": 0.172,
          "output": 0.572
        },
        "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
        "family": "Hy",
        "id": "tencent/hy3-preview",
        "last_updated": "2026-04-20",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Hy3 preview",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-20",
        "temperature": true,
        "tool_call": true
      },
      "volcengine/doubao-seed-1.8": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.0024,
          "input": 0.11,
          "output": 0.28
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "volcengine/doubao-seed-1.8",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-18",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed-1.8",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-12-18",
        "temperature": true,
        "tool_call": true
      },
      "volcengine/doubao-seed-2.0-code": {
        "attachment": true,
        "cost": {
          "input": 0.9,
          "output": 4.48
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "volcengine/doubao-seed-2.0-code",
        "knowledge": "2025-01-01",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 256000,
          "output": 32000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao Seed 2.0 Code",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "volcengine/doubao-seed-2.0-lite": {
        "attachment": true,
        "cost": {
          "cache_read": 0.02,
          "cache_write": 0.0024,
          "input": 0.09,
          "output": 0.51
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "volcengine/doubao-seed-2.0-lite",
        "knowledge": "2026-02-14",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed-2.0-lite",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-14",
        "temperature": true,
        "tool_call": true
      },
      "volcengine/doubao-seed-2.0-mini": {
        "attachment": true,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0.0024,
          "input": 0.03,
          "output": 0.28
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "volcengine/doubao-seed-2.0-mini",
        "knowledge": "2026-02-14",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed-2.0-mini",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-14",
        "temperature": true,
        "tool_call": true
      },
      "volcengine/doubao-seed-2.0-pro": {
        "attachment": true,
        "cost": {
          "cache_read": 0.09,
          "cache_write": 0.0024,
          "input": 0.45,
          "output": 2.24
        },
        "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
        "id": "volcengine/doubao-seed-2.0-pro",
        "knowledge": "2026-02-14",
        "last_updated": "2026-02-14",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed-2.0-pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-02-14",
        "temperature": true,
        "tool_call": true
      },
      "volcengine/doubao-seed-code": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.17,
          "output": 1.12
        },
        "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
        "id": "volcengine/doubao-seed-code",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-11",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Doubao-Seed-Code",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2025-11-11",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4": {
        "attachment": true,
        "cost": {
          "cache_read": 0.75,
          "input": 3,
          "output": 15
        },
        "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
        "id": "x-ai/grok-4",
        "knowledge": "2025-01-01",
        "last_updated": "2025-07-09",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "image",
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-07-09",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4-fast",
        "knowledge": "2025-01-01",
        "last_updated": "2025-09-19",
        "limit": {
          "context": 2000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-19",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.1-fast": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.1-fast",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 2000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.1-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "cache_read": 0.05,
          "input": 0.2,
          "output": 0.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.1-fast-non-reasoning",
        "knowledge": "2025-01-01",
        "last_updated": "2025-11-20",
        "limit": {
          "context": 2000000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.1 Fast Non Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2025-11-20",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.2-fast": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 9
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.2-fast",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.2 Fast",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.2-fast-non-reasoning": {
        "attachment": true,
        "cost": {
          "input": 3,
          "output": 9
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-4.2-fast-non-reasoning",
        "knowledge": "2025-08-31",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 2000000,
          "output": 30000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.2 Fast Non Reasoning",
        "open_weights": false,
        "reasoning": false,
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-4.3": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "context_over_200k": {
            "cache_read": 0.4,
            "cache_write": 0,
            "input": 2.5,
            "output": 5
          },
          "input": 1.25,
          "output": 2.5,
          "tiers": [
            {
              "cache_read": 0.4,
              "cache_write": 0,
              "input": 2.5,
              "output": 5,
              "tier": {
                "size": 200000,
                "type": "context"
              }
            }
          ]
        },
        "description": "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk",
        "family": "grok",
        "id": "x-ai/grok-4.3",
        "last_updated": "2026-04-17",
        "limit": {
          "context": 1000000,
          "output": 1000000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok 4.3",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "none",
              "low",
              "medium",
              "high"
            ]
          }
        ],
        "release_date": "2026-04-17",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-build-0.1": {
        "attachment": true,
        "cost": {
          "cache_read": 0.2,
          "input": 1,
          "output": 2
        },
        "description": "Fast Grok coding model tuned for agentic engineering and iterative edits",
        "family": "grok-build",
        "id": "x-ai/grok-build-0.1",
        "last_updated": "2026-04-16",
        "limit": {
          "context": 256000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Build 0.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "x-ai/grok-code-fast-1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.2,
          "output": 1.5
        },
        "description": "Fast Grok model for responsive chat, reasoning, and tool-assisted work",
        "id": "x-ai/grok-code-fast-1",
        "knowledge": "2025-01-01",
        "last_updated": "2025-08-26",
        "limit": {
          "context": 256000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "Grok Code Fast 1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-08-26",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.1,
          "output": 0.3
        },
        "description": "MiMo flash model for fast multimodal assistance and agent workflows",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-flash",
        "knowledge": "2024-12-01",
        "last_updated": "2026-02-04",
        "limit": {
          "context": 262144,
          "output": 65536
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-16",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-omni": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "input": 0.4,
          "output": 2
        },
        "description": "MiMo omni model for text, image, video, audio, and agents",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-omni",
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 265000,
          "output": 265000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Omni",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks",
        "family": "mimo",
        "id": "xiaomi/mimo-v2-pro",
        "knowledge": "2024-12",
        "last_updated": "2026-03-18",
        "limit": {
          "context": 1000000,
          "output": 256000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo V2 Pro",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-18",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5": {
        "attachment": true,
        "cost": {
          "cache_read": 0.08,
          "context_over_200k": {
            "cache_read": 0.16,
            "input": 0.8,
            "output": 4
          },
          "input": 0.4,
          "output": 2,
          "tiers": [
            {
              "cache_read": 0.16,
              "input": 0.8,
              "output": 4,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Open MiMo model for multimodal coding agents and long-context automation",
        "family": "mimo",
        "id": "xiaomi/mimo-v2.5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "audio",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "xiaomi/mimo-v2.5-pro": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "context_over_200k": {
            "cache_read": 0.4,
            "input": 2,
            "output": 6
          },
          "input": 1,
          "output": 3,
          "tiers": [
            {
              "cache_read": 0.4,
              "input": 2,
              "output": 6,
              "tier": {
                "size": 256000,
                "type": "context"
              }
            }
          ]
        },
        "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
        "family": "mimo",
        "id": "xiaomi/mimo-v2.5-pro",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2024-12",
        "last_updated": "2026-04-22",
        "limit": {
          "context": 1048576,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "MiMo-V2.5-Pro",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-22",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.07,
          "input": 0.35,
          "output": 1.54
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.5",
        "knowledge": "2025-01-01",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.02,
          "input": 0.11,
          "output": 0.56
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-4.5-air",
        "knowledge": "2025-01-01",
        "last_updated": "2025-07-25",
        "limit": {
          "context": 128000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.5 Air",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-25",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.07,
          "input": 0.35,
          "output": 1.54
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.6",
        "knowledge": "2025-01-01",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6v": {
        "attachment": true,
        "cost": {
          "cache_read": 0.03,
          "input": 0.14,
          "output": 0.42
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-4.6v",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6v-flash": {
        "attachment": true,
        "cost": {
          "cache_read": 0.0043,
          "input": 0.02,
          "output": 0.21
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-4.6v-flash",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V FlashX",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.6v-flash-free": {
        "attachment": true,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-4.6v-flash-free",
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.6V Flash (Free)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.06,
          "input": 0.28,
          "output": 1.14
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2025-12-23",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2025-12-23",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7-flash-free": {
        "attachment": false,
        "cost": {
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-4.7-flash-free",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 Flash (Free)",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "input": 0.07,
          "output": 0.42
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-4.7-flashx",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 64000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 4.7 FlashX",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.14,
          "input": 0.58,
          "output": 2.6
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-01-01",
        "last_updated": "2026-02-12",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-02-12",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5-turbo": {
        "attachment": true,
        "cost": {
          "input": 0.88,
          "output": 3.48
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "id": "z-ai/glm-5-turbo",
        "knowledge": "2025-01-01",
        "last_updated": "2026-03-20",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5 Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-03-20",
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.1903,
          "input": 0.8781,
          "output": 3.5126
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "id": "z-ai/glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-03",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-03",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "input": 1.4,
          "output": 4.5
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "z-ai/glm-5.2",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5.2-free": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "z-ai/glm-5.2-free",
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5.2 (Free)",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "z-ai/glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0.1743,
          "input": 0.726,
          "output": 3.1946
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "id": "z-ai/glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 128000
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM 5V Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "ZenMux",
    "npm": "@ai-sdk/openai-compatible"
  },
  "zhipuai": {
    "api": "https://open.bigmodel.cn/api/paas/v4",
    "doc": "https://docs.z.ai/guides/overview/pricing",
    "env": [
      "ZHIPU_API_KEY"
    ],
    "id": "zhipuai",
    "models": {
      "glm-4.5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
        "family": "glm",
        "id": "glm-4.5",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0.03,
          "cache_write": 0,
          "input": 0.2,
          "output": 1.1
        },
        "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "glm-4.5-flash",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.5v": {
        "attachment": true,
        "cost": {
          "input": 0.6,
          "output": 1.8
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.5v",
        "knowledge": "2025-04",
        "last_updated": "2025-08-11",
        "limit": {
          "context": 64000,
          "output": 16384
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-08-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
        "family": "glm",
        "id": "glm-4.6",
        "knowledge": "2025-04",
        "last_updated": "2025-09-30",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-09-30",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6v": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0.11,
          "cache_write": 0,
          "input": 0.6,
          "output": 2.2
        },
        "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flash": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
        "family": "glm-flash",
        "id": "glm-4.7-flash",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-Flash",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7-flashx": {
        "attachment": false,
        "cost": {
          "cache_read": 0.01,
          "cache_write": 0,
          "input": 0.07,
          "output": 0.4
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-flash",
        "id": "glm-4.7-flashx",
        "knowledge": "2025-04",
        "last_updated": "2026-01-19",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7-FlashX",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-01-19",
        "temperature": true,
        "tool_call": true
      },
      "glm-5": {
        "attachment": false,
        "cost": {
          "cache_read": 0.2,
          "cache_write": 0,
          "input": 1,
          "output": 3.2
        },
        "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
        "family": "glm",
        "id": "glm-5",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-02-11",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-02-11",
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0.26,
          "cache_write": 0,
          "input": 1.4,
          "output": 4.4
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 1.2,
          "cache_write": 0,
          "input": 5,
          "output": 22
        },
        "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
        "family": "glm",
        "id": "glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Zhipu AI",
    "npm": "@ai-sdk/openai-compatible"
  },
  "zhipuai-coding-plan": {
    "api": "https://open.bigmodel.cn/api/coding/paas/v4",
    "doc": "https://docs.bigmodel.cn/cn/coding-plan/overview",
    "env": [
      "ZHIPU_API_KEY"
    ],
    "id": "zhipuai-coding-plan",
    "models": {
      "glm-4.5-air": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm-air",
        "id": "glm-4.5-air",
        "knowledge": "2025-04",
        "last_updated": "2025-07-28",
        "limit": {
          "context": 131072,
          "output": 98304
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.5-Air",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-07-28",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.6v": {
        "attachment": true,
        "cost": {
          "input": 0.3,
          "output": 0.9
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-4.6v",
        "knowledge": "2025-04",
        "last_updated": "2025-12-08",
        "limit": {
          "context": 128000,
          "output": 32768
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.6V",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-08",
        "temperature": true,
        "tool_call": true
      },
      "glm-4.7": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-4.7",
        "interleaved": {
          "field": "reasoning_content"
        },
        "knowledge": "2025-04",
        "last_updated": "2025-12-22",
        "limit": {
          "context": 204800,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-4.7",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2025-12-22",
        "temperature": true,
        "tool_call": true
      },
      "glm-5-turbo": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
        "family": "glm",
        "id": "glm-5-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-16",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-16",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.1": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
        "family": "glm",
        "id": "glm-5.1",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-03-27",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.1",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-03-27",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5.2": {
        "attachment": false,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
        "family": "glm",
        "id": "glm-5.2",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-06-13",
        "limit": {
          "context": 1000000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5.2",
        "open_weights": true,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "effort",
            "values": [
              "high",
              "max"
            ]
          }
        ],
        "release_date": "2026-06-13",
        "structured_output": true,
        "temperature": true,
        "tool_call": true
      },
      "glm-5v-turbo": {
        "attachment": true,
        "cost": {
          "cache_read": 0,
          "cache_write": 0,
          "input": 0,
          "output": 0
        },
        "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
        "family": "glm",
        "id": "glm-5v-turbo",
        "interleaved": {
          "field": "reasoning_content"
        },
        "last_updated": "2026-04-01",
        "limit": {
          "context": 200000,
          "output": 131072
        },
        "modalities": {
          "input": [
            "text",
            "image",
            "video",
            "pdf"
          ],
          "output": [
            "text"
          ]
        },
        "name": "GLM-5V-Turbo",
        "open_weights": false,
        "reasoning": true,
        "reasoning_options": [
          {
            "type": "toggle"
          }
        ],
        "release_date": "2026-04-01",
        "temperature": true,
        "tool_call": true
      }
    },
    "name": "Zhipu AI Coding Plan",
    "npm": "@ai-sdk/openai-compatible"
  }
}
