[
{
"id": "claude-3-5-haiku-20241022",
"name": "Claude Haiku 3.5",
"provider": "anthropic",
"family": "claude-haiku",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 4,
"cache_read_input_per_million": 0.08,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 0.8,
"output": 4,
"cache_read": 0.08,
"cache_write": 1
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-07-31"
}
},
{
"id": "claude-3-5-haiku-latest",
"name": "Claude Haiku 3.5 (latest)",
"provider": "anthropic",
"family": "claude-haiku",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 4,
"cache_read_input_per_million": 0.08,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 0.8,
"output": 4,
"cache_read": 0.08,
"cache_write": 1
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-07-31"
}
},
{
"id": "claude-3-5-sonnet-20240620",
"name": "Claude Sonnet 3.5",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2024-06-20 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-04-30",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-06-20",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-04-30"
}
},
{
"id": "claude-3-5-sonnet-20241022",
"name": "Claude Sonnet 3.5 v2",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-04-30",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-04-30"
}
},
{
"id": "claude-3-7-sonnet-20250219",
"name": "Claude Sonnet 3.7",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2025-02-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2024-10-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-02-19",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2024-10-31"
}
},
{
"id": "claude-3-haiku-20240307",
"name": "Claude Haiku 3",
"provider": "anthropic",
"family": "claude-haiku",
"created_at": "2024-03-13 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 4096,
"knowledge_cutoff": "2023-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.25,
"cache_read_input_per_million": 0.03,
"cache_write_input_per_million": 0.3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-03-13",
"cost": {
"input": 0.25,
"output": 1.25,
"cache_read": 0.03,
"cache_write": 0.3
},
"limit": {
"context": 200000,
"output": 4096
},
"knowledge": "2023-08-31"
}
},
{
"id": "claude-3-opus-20240229",
"name": "Claude Opus 3",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2024-02-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 4096,
"knowledge_cutoff": "2023-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-02-29",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 4096
},
"knowledge": "2023-08-31"
}
},
{
"id": "claude-3-sonnet-20240229",
"name": "Claude Sonnet 3",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2024-03-04 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 4096,
"knowledge_cutoff": "2023-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 0.3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-03-04",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 0.3
},
"limit": {
"context": 200000,
"output": 4096
},
"knowledge": "2023-08-31"
}
},
{
"id": "claude-haiku-4-5",
"name": "Claude Haiku 4.5 (latest)",
"provider": "anthropic",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "claude-haiku-4-5-20251001",
"name": "Claude Haiku 4.5",
"provider": "anthropic",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "claude-opus-4-0",
"name": "Claude Opus 4 (latest)",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-1",
"name": "Claude Opus 4.1 (latest)",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-1-20250805",
"name": "Claude Opus 4.1",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-20250514",
"name": "Claude Opus 4",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-5",
"name": "Claude Opus 4.5 (latest)",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2025-11-24 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-24",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-5-20251101",
"name": "Claude Opus 4.5",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2025-11-01 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-01",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-6",
"name": "Claude Opus 4.6",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "claude-opus-4-7",
"name": "Claude Opus 4.7",
"provider": "anthropic",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "claude-sonnet-4-0",
"name": "Claude Sonnet 4 (latest)",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-sonnet-4-20250514",
"name": "Claude Sonnet 4",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-sonnet-4-5",
"name": "Claude Sonnet 4.5 (latest)",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "claude-sonnet-4-5-20250929",
"name": "Claude Sonnet 4.5",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "claude-sonnet-4-6",
"name": "Claude Sonnet 4.6",
"provider": "anthropic",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "anthropic",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-08-31"
}
},
{
"id": "AI21-Jamba-1.5-Large",
"name": "AI21-Jamba-1.5-Large",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "AI21-Jamba-1.5-Mini",
"name": "AI21-Jamba-1.5-Mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "AI21-Jamba-Instruct",
"name": "AI21-Jamba-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Codestral-2501-2",
"name": "Codestral-2501-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-command-r",
"name": "Cohere-command-r",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-command-r-08-2024",
"name": "Cohere-command-r-08-2024",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-command-r-plus",
"name": "Cohere-command-r-plus",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-command-r-plus-08-2024",
"name": "Cohere-command-r-plus-08-2024",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-embed-v3-english",
"name": "Cohere-embed-v3-english",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-embed-v3-multilingual",
"name": "Cohere-embed-v3-multilingual",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-rerank-v4.0-fast",
"name": "Cohere-rerank-v4.0-fast",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Cohere-rerank-v4.0-pro",
"name": "Cohere-rerank-v4.0-pro",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-R1",
"name": "DeepSeek-R1",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-R1-0528",
"name": "DeepSeek-R1-0528",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-V3",
"name": "DeepSeek-V3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-V3-0324",
"name": "DeepSeek-V3-0324",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-V3.1",
"name": "DeepSeek-V3.1",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-V3.2",
"name": "DeepSeek-V3.2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-V3.2-Speciale",
"name": "DeepSeek-V3.2-Speciale",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "DeepSeek-V4-Flash-2026-04-23",
"name": "DeepSeek-V4-Flash-2026-04-23",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "FLUX-1.1-pro",
"name": "FLUX-1.1-pro",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "FLUX.1-Kontext-pro",
"name": "FLUX.1-Kontext-pro",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "FLUX.2-pro",
"name": "FLUX.2-pro",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Kimi-K2-Thinking",
"name": "Kimi-K2-Thinking",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Kimi-K2.5",
"name": "Kimi-K2.5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Kimi-K2.6-2026-04-20",
"name": "Kimi-K2.6-2026-04-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.2-11B-Vision-Instruct",
"name": "Llama-3.2-11B-Vision-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.2-11B-Vision-Instruct-2",
"name": "Llama-3.2-11B-Vision-Instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.2-90B-Vision-Instruct",
"name": "Llama-3.2-90B-Vision-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.2-90B-Vision-Instruct-2",
"name": "Llama-3.2-90B-Vision-Instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.2-90B-Vision-Instruct-3",
"name": "Llama-3.2-90B-Vision-Instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.3-70B-Instruct",
"name": "Llama-3.3-70B-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.3-70B-Instruct-2",
"name": "Llama-3.3-70B-Instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.3-70B-Instruct-3",
"name": "Llama-3.3-70B-Instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.3-70B-Instruct-4",
"name": "Llama-3.3-70B-Instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.3-70B-Instruct-5",
"name": "Llama-3.3-70B-Instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-3.3-70B-Instruct-9",
"name": "Llama-3.3-70B-Instruct-9",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-4-Maverick-17B-128E-Instruct-FP8",
"name": "Llama-4-Maverick-17B-128E-Instruct-FP8",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Llama-4-Scout-17B-16E-Instruct",
"name": "Llama-4-Scout-17B-16E-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "MAI-DS-R1",
"name": "MAI-DS-R1",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "MAI-Image-2-2026-02-20",
"name": "MAI-Image-2-2026-02-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "MAI-Image-2e-2026-04-09",
"name": "MAI-Image-2e-2026-04-09",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-70B-Instruct-6",
"name": "Meta-Llama-3-70B-Instruct-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-70B-Instruct-7",
"name": "Meta-Llama-3-70B-Instruct-7",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-70B-Instruct-8",
"name": "Meta-Llama-3-70B-Instruct-8",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-70B-Instruct-9",
"name": "Meta-Llama-3-70B-Instruct-9",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-8B-Instruct-6",
"name": "Meta-Llama-3-8B-Instruct-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-8B-Instruct-7",
"name": "Meta-Llama-3-8B-Instruct-7",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-8B-Instruct-8",
"name": "Meta-Llama-3-8B-Instruct-8",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3-8B-Instruct-9",
"name": "Meta-Llama-3-8B-Instruct-9",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-405B-Instruct",
"name": "Meta-Llama-3.1-405B-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-70B-Instruct",
"name": "Meta-Llama-3.1-70B-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-70B-Instruct-2",
"name": "Meta-Llama-3.1-70B-Instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-70B-Instruct-3",
"name": "Meta-Llama-3.1-70B-Instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-70B-Instruct-4",
"name": "Meta-Llama-3.1-70B-Instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-8B-Instruct",
"name": "Meta-Llama-3.1-8B-Instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-8B-Instruct-2",
"name": "Meta-Llama-3.1-8B-Instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-8B-Instruct-3",
"name": "Meta-Llama-3.1-8B-Instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-8B-Instruct-4",
"name": "Meta-Llama-3.1-8B-Instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Meta-Llama-3.1-8B-Instruct-5",
"name": "Meta-Llama-3.1-8B-Instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Ministral-3B",
"name": "Ministral-3B",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Mistral-Large-2411-2",
"name": "Mistral-Large-2411-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Mistral-Large-3",
"name": "Mistral-Large-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Mistral-Nemo",
"name": "Mistral-Nemo",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Mistral-large",
"name": "Mistral-large",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Mistral-large-2407",
"name": "Mistral-large-2407",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Mistral-small",
"name": "Mistral-small",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-128k-instruct-3",
"name": "Phi-3-medium-128k-instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-128k-instruct-4",
"name": "Phi-3-medium-128k-instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-128k-instruct-5",
"name": "Phi-3-medium-128k-instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-128k-instruct-6",
"name": "Phi-3-medium-128k-instruct-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-128k-instruct-7",
"name": "Phi-3-medium-128k-instruct-7",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-4k-instruct-3",
"name": "Phi-3-medium-4k-instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-4k-instruct-4",
"name": "Phi-3-medium-4k-instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-4k-instruct-5",
"name": "Phi-3-medium-4k-instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-medium-4k-instruct-6",
"name": "Phi-3-medium-4k-instruct-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-128k-instruct-10",
"name": "Phi-3-mini-128k-instruct-10",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-128k-instruct-11",
"name": "Phi-3-mini-128k-instruct-11",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-128k-instruct-12",
"name": "Phi-3-mini-128k-instruct-12",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-128k-instruct-13",
"name": "Phi-3-mini-128k-instruct-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-4k-instruct-10",
"name": "Phi-3-mini-4k-instruct-10",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-4k-instruct-11",
"name": "Phi-3-mini-4k-instruct-11",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-4k-instruct-13",
"name": "Phi-3-mini-4k-instruct-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-4k-instruct-14",
"name": "Phi-3-mini-4k-instruct-14",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-mini-4k-instruct-15",
"name": "Phi-3-mini-4k-instruct-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-small-128k-instruct-3",
"name": "Phi-3-small-128k-instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-small-128k-instruct-4",
"name": "Phi-3-small-128k-instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-small-128k-instruct-5",
"name": "Phi-3-small-128k-instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-small-8k-instruct-3",
"name": "Phi-3-small-8k-instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-small-8k-instruct-4",
"name": "Phi-3-small-8k-instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3-small-8k-instruct-5",
"name": "Phi-3-small-8k-instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-MoE-instruct-2",
"name": "Phi-3.5-MoE-instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-MoE-instruct-3",
"name": "Phi-3.5-MoE-instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-MoE-instruct-4",
"name": "Phi-3.5-MoE-instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-MoE-instruct-5",
"name": "Phi-3.5-MoE-instruct-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-mini-instruct",
"name": "Phi-3.5-mini-instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-mini-instruct-2",
"name": "Phi-3.5-mini-instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-mini-instruct-3",
"name": "Phi-3.5-mini-instruct-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-mini-instruct-4",
"name": "Phi-3.5-mini-instruct-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-mini-instruct-6",
"name": "Phi-3.5-mini-instruct-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-vision-instruct",
"name": "Phi-3.5-vision-instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-3.5-vision-instruct-2",
"name": "Phi-3.5-vision-instruct-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-2",
"name": "Phi-4-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-3",
"name": "Phi-4-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-4",
"name": "Phi-4-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-5",
"name": "Phi-4-5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-6",
"name": "Phi-4-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-7",
"name": "Phi-4-7",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-mini-instruct",
"name": "Phi-4-mini-instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-mini-reasoning",
"name": "Phi-4-mini-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-multimodal-instruct",
"name": "Phi-4-multimodal-instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Phi-4-reasoning",
"name": "Phi-4-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Stable-Diffusion-3.5-Large",
"name": "Stable-Diffusion-3.5-Large",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Stable-Image-Core",
"name": "Stable-Image-Core",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "Stable-Image-Ultra",
"name": "Stable-Image-Ultra",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "ada",
"name": "ada",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "aoai-sora",
"name": "aoai-sora",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "aoai-sora-2025-02-28",
"name": "aoai-sora-2025-02-28",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "babbage",
"name": "babbage",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 0.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-haiku-4-5-20251001",
"name": "claude-haiku-4-5-20251001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-opus-4-1-20250805",
"name": "claude-opus-4-1-20250805",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-opus-4-5-20251101",
"name": "claude-opus-4-5-20251101",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-opus-4-6",
"name": "claude-opus-4-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-opus-4-7",
"name": "claude-opus-4-7",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-sonnet-4-5-20250929",
"name": "claude-sonnet-4-5-20250929",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "claude-sonnet-4-6",
"name": "claude-sonnet-4-6",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-cushman-001",
"name": "code-cushman-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-cushman-fine-tune-002",
"name": "code-cushman-fine-tune-002",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-davinci-002",
"name": "code-davinci-002",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-davinci-fine-tune-002",
"name": "code-davinci-fine-tune-002",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-search-ada-code-001",
"name": "code-search-ada-code-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-search-ada-text-001",
"name": "code-search-ada-text-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-search-babbage-code-001",
"name": "code-search-babbage-code-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "code-search-babbage-text-001",
"name": "code-search-babbage-text-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "codex-mini-2025-05-16",
"name": "codex-mini-2025-05-16",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "cohere-command-a",
"name": "cohere-command-a",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "computer-use-preview-2025-04-15",
"name": "computer-use-preview-2025-04-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "curie",
"name": "curie",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "dall-e-2",
"name": "dall-e-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "dall-e-2-2.0",
"name": "dall-e-2-2.0",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "dall-e-3",
"name": "dall-e-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "dall-e-3-3.0",
"name": "dall-e-3-3.0",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "davinci",
"name": "davinci",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 2.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "embed-v-4-0",
"name": "embed-v-4-0",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo",
"name": "gpt-35-turbo",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-0125",
"name": "gpt-35-turbo-0125",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-0301",
"name": "gpt-35-turbo-0301",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-0613",
"name": "gpt-35-turbo-0613",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-1106",
"name": "gpt-35-turbo-1106",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-16k",
"name": "gpt-35-turbo-16k",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-16k-0613",
"name": "gpt-35-turbo-16k-0613",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-instruct",
"name": "gpt-35-turbo-instruct",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-35-turbo-instruct-0914",
"name": "gpt-35-turbo-instruct-0914",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4",
"name": "gpt-4",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-0125-Preview",
"name": "gpt-4-0125-Preview",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-0314",
"name": "gpt-4-0314",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-0613",
"name": "gpt-4-0613",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-1106-Preview",
"name": "gpt-4-1106-Preview",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-32k",
"name": "gpt-4-32k",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-32k-0314",
"name": "gpt-4-32k-0314",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-32k-0613",
"name": "gpt-4-32k-0613",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-turbo-2024-04-09",
"name": "gpt-4-turbo-2024-04-09",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-turbo-jp",
"name": "gpt-4-turbo-jp",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4-vision-preview",
"name": "gpt-4-vision-preview",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1",
"name": "gpt-4.1",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1-2025-04-14",
"name": "gpt-4.1-2025-04-14",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1-2025-04-14-text",
"name": "gpt-4.1-2025-04-14-text",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1-mini",
"name": "gpt-4.1-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 1.6,
"cache_read_input_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1-mini-2025-04-14",
"name": "gpt-4.1-mini-2025-04-14",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 1.6,
"cache_read_input_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1-nano",
"name": "gpt-4.1-nano",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4.1-nano-2025-04-14",
"name": "gpt-4.1-nano-2025-04-14",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o",
"name": "gpt-4o",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-2024-05-13",
"name": "gpt-4o-2024-05-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-2024-08-06",
"name": "gpt-4o-2024-08-06",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-2024-11-20",
"name": "gpt-4o-2024-11-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-audio-mai",
"name": "gpt-4o-audio-mai",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-audio-preview-2024-10-01",
"name": "gpt-4o-audio-preview-2024-10-01",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-audio-preview-2024-12-17",
"name": "gpt-4o-audio-preview-2024-12-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-audio-preview-2025-06-03",
"name": "gpt-4o-audio-preview-2025-06-03",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-canvas-2024-09-25",
"name": "gpt-4o-canvas-2024-09-25",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini",
"name": "gpt-4o-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-2024-07-18",
"name": "gpt-4o-mini-2024-07-18",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-audio-preview-2024-12-17",
"name": "gpt-4o-mini-audio-preview-2024-12-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-realtime-preview-2024-12-17",
"name": "gpt-4o-mini-realtime-preview-2024-12-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-transcribe",
"name": "gpt-4o-mini-transcribe",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 16000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-transcribe-2025-03-20",
"name": "gpt-4o-mini-transcribe-2025-03-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 16000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-transcribe-2025-12-15",
"name": "gpt-4o-mini-transcribe-2025-12-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 16000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-tts",
"name": "gpt-4o-mini-tts",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 12.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-tts-2025-03-20",
"name": "gpt-4o-mini-tts-2025-03-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 12.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-mini-tts-2025-12-15",
"name": "gpt-4o-mini-tts-2025-12-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 12.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-realtime-preview",
"name": "gpt-4o-realtime-preview",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 20.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-realtime-preview-2024-12-17",
"name": "gpt-4o-realtime-preview-2024-12-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 20.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-realtime-preview-2025-06-03",
"name": "gpt-4o-realtime-preview-2025-06-03",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 20.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-transcribe",
"name": "gpt-4o-transcribe",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-transcribe-2025-03-20",
"name": "gpt-4o-transcribe-2025-03-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-transcribe-diarize",
"name": "gpt-4o-transcribe-diarize",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-4o-transcribe-diarize-2025-10-15",
"name": "gpt-4o-transcribe-diarize-2025-10-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-2025-08-07",
"name": "gpt-5-2025-08-07",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-chat-2025-08-07",
"name": "gpt-5-chat-2025-08-07",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-chat-2025-08-15",
"name": "gpt-5-chat-2025-08-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-chat-2025-10-03",
"name": "gpt-5-chat-2025-10-03",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-codex-2025-09-15",
"name": "gpt-5-codex-2025-09-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-mini-2025-08-07",
"name": "gpt-5-mini-2025-08-07",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-mini-2025-08-07-lite",
"name": "gpt-5-mini-2025-08-07-lite",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-mini-lite-2025-08-07",
"name": "gpt-5-mini-lite-2025-08-07",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-nano-2025-08-07",
"name": "gpt-5-nano-2025-08-07",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.005
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5-pro-2025-10-06",
"name": "gpt-5-pro-2025-10-06",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.1",
"name": "gpt-5.1",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.1-2025-11-13",
"name": "gpt-5.1-2025-11-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.1-chat-2025-11-13",
"name": "gpt-5.1-chat-2025-11-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.1-codex-2025-11-13",
"name": "gpt-5.1-codex-2025-11-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.1-codex-max-2025-12-04",
"name": "gpt-5.1-codex-max-2025-12-04",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.1-codex-mini-2025-11-13",
"name": "gpt-5.1-codex-mini-2025-11-13",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.2-2025-12-11",
"name": "gpt-5.2-2025-12-11",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.2-chat-2025-12-11",
"name": "gpt-5.2-chat-2025-12-11",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.2-chat-2026-02-10",
"name": "gpt-5.2-chat-2026-02-10",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.2-codex-2026-01-14",
"name": "gpt-5.2-codex-2026-01-14",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.3-chat-2026-03-03",
"name": "gpt-5.3-chat-2026-03-03",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.3-codex-2026-02-20",
"name": "gpt-5.3-codex-2026-02-20",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.3-codex-2026-02-24",
"name": "gpt-5.3-codex-2026-02-24",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.4-2026-03-05",
"name": "gpt-5.4-2026-03-05",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.4-mini-2026-03-17",
"name": "gpt-5.4-mini-2026-03-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.4-nano-2026-03-17",
"name": "gpt-5.4-nano-2026-03-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.005
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.4-pro-2026-03-05",
"name": "gpt-5.4-pro-2026-03-05",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-5.5-2026-04-24",
"name": "gpt-5.5-2026-04-24",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-audio-1.5-2026-02-23",
"name": "gpt-audio-1.5-2026-02-23",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-audio-2025-08-28",
"name": "gpt-audio-2025-08-28",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-audio-mini-2025-10-06",
"name": "gpt-audio-mini-2025-10-06",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-chat-latest-2026-05-05",
"name": "gpt-chat-latest-2026-05-05",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-image-1",
"name": "gpt-image-1",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"cache_read_input_per_million": 1.25
}
},
"images": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 40.0,
"cache_read_input_per_million": 2.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-image-1-2025-04-15",
"name": "gpt-image-1-2025-04-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"cache_read_input_per_million": 1.25
}
},
"images": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 40.0,
"cache_read_input_per_million": 2.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-image-1-mini",
"name": "gpt-image-1-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"cache_read_input_per_million": 0.2
}
},
"images": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-image-1-mini-2025-10-06",
"name": "gpt-image-1-mini-2025-10-06",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"cache_read_input_per_million": 0.2
}
},
"images": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-image-1.5",
"name": "gpt-image-1.5",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"cache_read_input_per_million": 1.25
}
},
"images": {
"standard": {
"input_per_million": 8.0,
"output_per_million": 32.0,
"cache_read_input_per_million": 2.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-image-1.5-2025-12-16",
"name": "gpt-image-1.5-2025-12-16",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"cache_read_input_per_million": 1.25
}
},
"images": {
"standard": {
"input_per_million": 8.0,
"output_per_million": 32.0,
"cache_read_input_per_million": 2.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-oss-120b",
"name": "gpt-oss-120b",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-oss-20b-11",
"name": "gpt-oss-20b-11",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-realtime-1.5-2026-02-23",
"name": "gpt-realtime-1.5-2026-02-23",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-realtime-2025-08-28",
"name": "gpt-realtime-2025-08-28",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-realtime-mini",
"name": "gpt-realtime-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-realtime-mini-2025-10-06",
"name": "gpt-realtime-mini-2025-10-06",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "gpt-realtime-mini-2025-12-15",
"name": "gpt-realtime-mini-2025-12-15",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-3",
"name": "grok-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-3-mini",
"name": "grok-3-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-4-1-fast-non-reasoning",
"name": "grok-4-1-fast-non-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-4-1-fast-reasoning",
"name": "grok-4-1-fast-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-4-20-non-reasoning",
"name": "grok-4-20-non-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-4-20-reasoning",
"name": "grok-4-20-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-4-fast-non-reasoning",
"name": "grok-4-fast-non-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "grok-4-fast-reasoning",
"name": "grok-4-fast-reasoning",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "jais-30b-chat",
"name": "jais-30b-chat",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "jais-30b-chat-2",
"name": "jais-30b-chat-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "jais-30b-chat-3",
"name": "jais-30b-chat-3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "mistral-document-ai-2505",
"name": "mistral-document-ai-2505",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "mistral-document-ai-2512",
"name": "mistral-document-ai-2512",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "mistral-medium-2505",
"name": "mistral-medium-2505",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "mistral-small-2503",
"name": "mistral-small-2503",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "model-router",
"name": "model-router",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "model-router-2025-05-19",
"name": "model-router-2025-05-19",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "model-router-2025-08-07",
"name": "model-router-2025-08-07",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "model-router-2025-11-18",
"name": "model-router-2025-11-18",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o1-2024-12-17",
"name": "o1-2024-12-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15.0,
"output_per_million": 60.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o1-mini-2024-09-12",
"name": "o1-mini-2024-09-12",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o1-pro",
"name": "o1-pro",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 150.0,
"output_per_million": 600.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o1-pro-2025-03-19",
"name": "o1-pro-2025-03-19",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 150.0,
"output_per_million": 600.0
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o3-deep-research-2025-06-26",
"name": "o3-deep-research-2025-06-26",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o3-deep-research-2025-06-26-ev3",
"name": "o3-deep-research-2025-06-26-ev3",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o3-mini",
"name": "o3-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o3-mini-2025-01-31",
"name": "o3-mini-2025-01-31",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o3-mini-alpha",
"name": "o3-mini-alpha",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o3-mini-alpha-2024-12-17",
"name": "o3-mini-alpha-2024-12-17",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o4-mini",
"name": "o4-mini",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "o4-mini-2025-04-16",
"name": "o4-mini-2025-04-16",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "qwen-3-32b",
"name": "qwen-3-32b",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "qwen3-32b",
"name": "qwen3-32b",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "sora",
"name": "sora",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "sora-2",
"name": "sora-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "sora-2-2025-10-06",
"name": "sora-2-2025-10-06",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "sora-2-2025-12-08",
"name": "sora-2-2025-12-08",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "sora-2025-05-02",
"name": "sora-2025-05-02",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-ada-001",
"name": "text-ada-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-babbage-001",
"name": "text-babbage-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-curie-001",
"name": "text-curie-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-davinci-001",
"name": "text-davinci-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-davinci-002",
"name": "text-davinci-002",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-davinci-003",
"name": "text-davinci-003",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-davinci-fine-tune-002",
"name": "text-davinci-fine-tune-002",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-embedding-3-large",
"name": "text-embedding-3-large",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.13,
"output_per_million": 0.13
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-embedding-3-small",
"name": "text-embedding-3-small",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.02,
"output_per_million": 0.02
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-embedding-ada-002",
"name": "text-embedding-ada-002",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-embedding-ada-002-2",
"name": "text-embedding-ada-002-2",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-ada-doc-001",
"name": "text-search-ada-doc-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-ada-query-001",
"name": "text-search-ada-query-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-babbage-doc-001",
"name": "text-search-babbage-doc-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-babbage-query-001",
"name": "text-search-babbage-query-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-curie-doc-001",
"name": "text-search-curie-doc-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-curie-query-001",
"name": "text-search-curie-query-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-davinci-doc-001",
"name": "text-search-davinci-doc-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-search-davinci-query-001",
"name": "text-search-davinci-query-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-similarity-ada-001",
"name": "text-similarity-ada-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-similarity-babbage-001",
"name": "text-similarity-babbage-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-similarity-curie-001",
"name": "text-similarity-curie-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "text-similarity-davinci-001",
"name": "text-similarity-davinci-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "whisper",
"name": "whisper",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.006,
"output_per_million": 0.006
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "whisper-001",
"name": "whisper-001",
"provider": "azure",
"family": null,
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.006,
"output_per_million": 0.006
}
}
},
"metadata": {
"object": "model",
"owned_by": null
}
},
{
"id": "amazon.nova-2-lite-v1:0",
"name": "Nova 2 Lite",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.33,
"output_per_million": 2.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.33,
"output": 2.75
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "amazon.nova-2-sonic-v1:0",
"name": "Nova 2 Sonic",
"provider": "bedrock",
"family": "Nova",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"audio"
],
"output": [
"audio",
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-2-sonic-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "amazon.nova-lite-v1:0",
"name": "Nova Lite",
"provider": "bedrock",
"family": "nova-lite",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 300000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.24,
"cache_read_input_per_million": 0.015
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 0.06,
"output": 0.24,
"cache_read": 0.015
},
"limit": {
"context": 300000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.nova-micro-v1:0",
"name": "Nova Micro",
"provider": "bedrock",
"family": "nova-micro",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.035,
"output_per_million": 0.14,
"cache_read_input_per_million": 0.00875
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 0.035,
"output": 0.14,
"cache_read": 0.00875
},
"limit": {
"context": 128000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.nova-premier-v1:0",
"name": "Nova Premier",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 12.5
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 2.5,
"output": 12.5
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.nova-premier-v1:0:1000k",
"name": "Nova Premier",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 12.5
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-premier-v1:0:1000k",
"inference_types": [],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 2.5,
"output": 12.5
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.nova-premier-v1:0:20k",
"name": "Nova Premier",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 20000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 12.5
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-premier-v1:0:20k",
"inference_types": [],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 2.5,
"output": 12.5
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.nova-premier-v1:0:8k",
"name": "Nova Premier",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 12.5
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-premier-v1:0:8k",
"inference_types": [],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 2.5,
"output": 12.5
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.nova-premier-v1:0:mm",
"name": "Nova Premier",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-premier-v1:0:mm",
"inference_types": [],
"converse": {}
}
},
{
"id": "amazon.nova-pro-v1:0",
"name": "Nova Pro",
"provider": "bedrock",
"family": "nova-pro",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 300000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 3.2,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 0.8,
"output": 3.2,
"cache_read": 0.2
},
"limit": {
"context": 300000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "amazon.rerank-v1:0",
"name": "Rerank 1.0",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.rerank-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "amazon.titan-embed-g1-text-02",
"name": "Titan Text Embeddings v2",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-embed-g1-text-02",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "amazon.titan-embed-image-v1",
"name": "Titan Multimodal Embeddings G1",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-embed-image-v1",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "amazon.titan-embed-image-v1:0",
"name": "Titan Multimodal Embeddings G1",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-embed-image-v1:0",
"inference_types": [
"PROVISIONED"
],
"converse": {}
}
},
{
"id": "amazon.titan-embed-text-v1",
"name": "Titan Embeddings G1 - Text",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-embed-text-v1",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "amazon.titan-embed-text-v1:2:8k",
"name": "Titan Embeddings G1 - Text",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-embed-text-v1:2:8k",
"inference_types": [
"PROVISIONED"
],
"converse": {}
}
},
{
"id": "amazon.titan-embed-text-v2:0",
"name": "Titan Text Embeddings V2",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-embed-text-v2:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "amazon.titan-image-generator-v2:0",
"name": "Titan Image Generator G1 v2",
"provider": "bedrock",
"family": "amazon",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.titan-image-generator-v2:0",
"inference_types": [
"PROVISIONED",
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "anthropic.claude-3-5-haiku-20241022-v1:0",
"name": "Claude Haiku 3.5",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 4,
"cache_read_input_per_million": 0.08,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-5-haiku-20241022-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 0.8,
"output": 4,
"cache_read": 0.08,
"cache_write": 1
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-07-31"
}
},
{
"id": "anthropic.claude-3-5-sonnet-20240620-v1:0",
"name": "Claude Sonnet 3.5",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2024-06-20 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-04-30",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-06-20",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-04-30"
}
},
{
"id": "anthropic.claude-3-5-sonnet-20241022-v2:0",
"name": "Claude Sonnet 3.5 v2",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "anthropic.claude-3-7-sonnet-20250219-v1:0",
"name": "Claude Sonnet 3.7",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-02-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-02-19",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "anthropic.claude-3-haiku-20240307-v1:0",
"name": "Claude Haiku 3",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2024-03-13 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-haiku-20240307-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-03-13",
"cost": {
"input": 0.25,
"output": 1.25
},
"limit": {
"context": 200000,
"output": 4096
},
"knowledge": "2024-02"
}
},
{
"id": "anthropic.claude-3-haiku-20240307-v1:0:200k",
"name": "Claude Haiku 3",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2024-03-13 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-haiku-20240307-v1:0:200k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-03-13",
"cost": {
"input": 0.25,
"output": 1.25
},
"limit": {
"context": 200000,
"output": 4096
},
"knowledge": "2024-02"
}
},
{
"id": "anthropic.claude-3-haiku-20240307-v1:0:48k",
"name": "Claude Haiku 3",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2024-03-13 00:00:00 UTC",
"context_window": 48000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-haiku-20240307-v1:0:48k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-03-13",
"cost": {
"input": 0.25,
"output": 1.25
},
"limit": {
"context": 200000,
"output": 4096
},
"knowledge": "2024-02"
}
},
{
"id": "anthropic.claude-3-sonnet-20240229-v1:0",
"name": "Claude 3 Sonnet",
"provider": "bedrock",
"family": "anthropic",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-sonnet-20240229-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "anthropic.claude-3-sonnet-20240229-v1:0:200k",
"name": "Claude 3 Sonnet",
"provider": "bedrock",
"family": "anthropic",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-sonnet-20240229-v1:0:200k",
"inference_types": [
"PROVISIONED"
],
"converse": {}
}
},
{
"id": "anthropic.claude-3-sonnet-20240229-v1:0:28k",
"name": "Claude 3 Sonnet",
"provider": "bedrock",
"family": "anthropic",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-3-sonnet-20240229-v1:0:28k",
"inference_types": [
"PROVISIONED"
],
"converse": {}
}
},
{
"id": "anthropic.claude-haiku-4-5-20251001-v1:0",
"name": "Claude Haiku 4.5",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "anthropic.claude-opus-4-1-20250805-v1:0",
"name": "Claude Opus 4.1",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic.claude-opus-4-20250514-v1:0",
"name": "Claude Opus 4",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic.claude-opus-4-5-20251101-v1:0",
"name": "Claude Opus 4.5",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-11-24 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-01",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic.claude-opus-4-6-v1",
"name": "Claude Opus 4.6",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "anthropic.claude-opus-4-7",
"name": "Claude Opus 4.7",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "anthropic.claude-sonnet-4-20250514-v1:0",
"name": "Claude Sonnet 4",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic.claude-sonnet-4-5-20250929-v1:0",
"name": "Claude Sonnet 4.5",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "anthropic.claude-sonnet-4-6",
"name": "Claude Sonnet 4.6",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-08-31"
}
},
{
"id": "au.anthropic.claude-opus-4-6-v1",
"name": "AU Anthropic Claude Opus 4.6",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 16.5,
"output_per_million": 82.5,
"cache_read_input_per_million": 1.65,
"cache_write_input_per_million": 20.625
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-05",
"cost": {
"input": 16.5,
"output": 82.5,
"cache_read": 1.65,
"cache_write": 20.625
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05"
}
},
{
"id": "au.anthropic.claude-sonnet-4-6",
"name": "AU Anthropic Claude Sonnet 4.6",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.3,
"output_per_million": 16.5,
"cache_read_input_per_million": 0.33,
"cache_write_input_per_million": 4.125
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-17",
"cost": {
"input": 3.3,
"output": 16.5,
"cache_read": 0.33,
"cache_write": 4.125
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-08"
}
},
{
"id": "cohere.command-r-plus-v1:0",
"name": "Command R+",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.command-r-plus-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "cohere.command-r-v1:0",
"name": "Command R",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.command-r-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "cohere.embed-english-v3",
"name": "Embed English",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.embed-english-v3",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "cohere.embed-english-v3:0:512",
"name": "Embed English",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.embed-english-v3:0:512",
"inference_types": [
"PROVISIONED"
],
"converse": {}
}
},
{
"id": "cohere.embed-multilingual-v3",
"name": "Embed Multilingual",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.embed-multilingual-v3",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "cohere.embed-multilingual-v3:0:512",
"name": "Embed Multilingual",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.embed-multilingual-v3:0:512",
"inference_types": [
"PROVISIONED"
],
"converse": {}
}
},
{
"id": "cohere.rerank-v3-5:0",
"name": "Rerank 3.5",
"provider": "bedrock",
"family": "cohere",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.rerank-v3-5:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "deepseek.r1-v1:0",
"name": "DeepSeek-R1",
"provider": "bedrock",
"family": "deepseek-thinking",
"created_at": "2025-01-20 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.35,
"output_per_million": 5.4
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-05-29",
"cost": {
"input": 1.35,
"output": 5.4
},
"limit": {
"context": 128000,
"output": 32768
},
"knowledge": "2024-07"
}
},
{
"id": "deepseek.v3-v1:0",
"name": "DeepSeek-V3.1",
"provider": "bedrock",
"family": "deepseek",
"created_at": "2025-09-18 00:00:00 UTC",
"context_window": 163840,
"max_output_tokens": 81920,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.58,
"output_per_million": 1.68
}
}
},
"metadata": {
"provider_name": "DeepSeek",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/deepseek.v3-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 163840,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-18",
"cost": {
"input": 0.58,
"output": 1.68
},
"limit": {
"context": 163840,
"output": 81920
},
"knowledge": "2024-07"
}
},
{
"id": "deepseek.v3.2",
"name": "DeepSeek-V3.2",
"provider": "bedrock",
"family": "deepseek",
"created_at": "2026-02-06 00:00:00 UTC",
"context_window": 163840,
"max_output_tokens": 81920,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.62,
"output_per_million": 1.85
}
}
},
"metadata": {
"provider_name": "DeepSeek",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/deepseek.v3.2",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 163840,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-06",
"cost": {
"input": 0.62,
"output": 1.85
},
"limit": {
"context": 163840,
"output": 81920
},
"knowledge": "2024-07"
}
},
{
"id": "eu.anthropic.claude-haiku-4-5-20251001-v1:0",
"name": "Claude Haiku 4.5 (EU)",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "eu.anthropic.claude-opus-4-5-20251101-v1:0",
"name": "Claude Opus 4.5 (EU)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-11-24 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-01",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "eu.anthropic.claude-opus-4-6-v1",
"name": "Claude Opus 4.6 (EU)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "eu.anthropic.claude-opus-4-7",
"name": "Claude Opus 4.7 (EU)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "eu.anthropic.claude-sonnet-4-20250514-v1:0",
"name": "Claude Sonnet 4 (EU)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "eu.anthropic.claude-sonnet-4-5-20250929-v1:0",
"name": "Claude Sonnet 4.5 (EU)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "eu.anthropic.claude-sonnet-4-6",
"name": "Claude Sonnet 4.6 (EU)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-08-31"
}
},
{
"id": "global.anthropic.claude-haiku-4-5-20251001-v1:0",
"name": "Claude Haiku 4.5 (Global)",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "global.anthropic.claude-opus-4-5-20251101-v1:0",
"name": "Claude Opus 4.5 (Global)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-11-24 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-01",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "global.anthropic.claude-opus-4-6-v1",
"name": "Claude Opus 4.6 (Global)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "global.anthropic.claude-opus-4-7",
"name": "Claude Opus 4.7 (Global)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "global.anthropic.claude-sonnet-4-20250514-v1:0",
"name": "Claude Sonnet 4 (Global)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "global.anthropic.claude-sonnet-4-5-20250929-v1:0",
"name": "Claude Sonnet 4.5 (Global)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "global.anthropic.claude-sonnet-4-6",
"name": "Claude Sonnet 4.6 (Global)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-08-31"
}
},
{
"id": "google.gemma-3-12b-it",
"name": "Google Gemma 3 12B",
"provider": "bedrock",
"family": "gemma",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"vision",
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.049999999999999996,
"output_per_million": 0.09999999999999999
}
}
},
"metadata": {
"provider_name": "Google",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/google.gemma-3-12b-it",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.049999999999999996,
"output": 0.09999999999999999
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-12"
}
},
{
"id": "google.gemma-3-27b-it",
"name": "Google Gemma 3 27B Instruct",
"provider": "bedrock",
"family": "gemma",
"created_at": "2025-07-27 00:00:00 UTC",
"context_window": 202752,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.12,
"output_per_million": 0.2
}
}
},
"metadata": {
"provider_name": "Google",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/google.gemma-3-27b-it",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-07-27",
"cost": {
"input": 0.12,
"output": 0.2
},
"limit": {
"context": 202752,
"output": 8192
},
"knowledge": "2025-07"
}
},
{
"id": "google.gemma-3-4b-it",
"name": "Gemma 3 4B IT",
"provider": "bedrock",
"family": "gemma",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.08
}
}
},
"metadata": {
"provider_name": "Google",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/google.gemma-3-4b-it",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.04,
"output": 0.08
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "luma.ray-v2:0",
"name": "Ray v2",
"provider": "bedrock",
"family": "luma ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"video"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Luma AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/luma.ray-v2:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "meta.llama3-1-405b-instruct-v1:0",
"name": "Llama 3.1 405B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.4,
"output_per_million": 2.4
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-1-405b-instruct-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-07-23",
"cost": {
"input": 2.4,
"output": 2.4
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-1-70b-instruct-v1:0",
"name": "Llama 3.1 70B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-1-70b-instruct-v1:0",
"inference_types": [
"ON_DEMAND",
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-07-23",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-1-70b-instruct-v1:0:128k",
"name": "Llama 3.1 70B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-1-70b-instruct-v1:0:128k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-07-23",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-1-8b-instruct-v1:0",
"name": "Llama 3.1 8B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.22,
"output_per_million": 0.22
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-1-8b-instruct-v1:0",
"inference_types": [
"ON_DEMAND",
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-07-23",
"cost": {
"input": 0.22,
"output": 0.22
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-1-8b-instruct-v1:0:128k",
"name": "Llama 3.1 8B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.22,
"output_per_million": 0.22
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-1-8b-instruct-v1:0:128k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-07-23",
"cost": {
"input": 0.22,
"output": 0.22
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-11b-instruct-v1:0",
"name": "Llama 3.2 11B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.16,
"output_per_million": 0.16
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.16,
"output": 0.16
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-11b-instruct-v1:0:128k",
"name": "Llama 3.2 11B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.16,
"output_per_million": 0.16
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-11b-instruct-v1:0:128k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.16,
"output": 0.16
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-1b-instruct-v1:0",
"name": "Llama 3.2 1B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 131000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.1,
"output": 0.1
},
"limit": {
"context": 131000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-1b-instruct-v1:0:128k",
"name": "Llama 3.2 1B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-1b-instruct-v1:0:128k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.1,
"output": 0.1
},
"limit": {
"context": 131000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-3b-instruct-v1:0",
"name": "Llama 3.2 3B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 131000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.15,
"output": 0.15
},
"limit": {
"context": 131000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-3b-instruct-v1:0:128k",
"name": "Llama 3.2 3B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-3b-instruct-v1:0:128k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.15,
"output": 0.15
},
"limit": {
"context": 131000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-90b-instruct-v1:0",
"name": "Llama 3.2 90B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-2-90b-instruct-v1:0:128k",
"name": "Llama 3.2 90B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-90b-instruct-v1:0:128k",
"inference_types": [
"PROVISIONED"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-3-70b-instruct-v1:0",
"name": "Llama 3.3 70B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-12-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-06",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-3-70b-instruct-v1:0:128k",
"name": "Llama 3.3 70B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-12-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-3-70b-instruct-v1:0:128k",
"inference_types": [],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-06",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "meta.llama3-70b-instruct-v1:0",
"name": "Llama 3 70B Instruct",
"provider": "bedrock",
"family": "meta",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-70b-instruct-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "meta.llama3-8b-instruct-v1:0",
"name": "Llama 3 8B Instruct",
"provider": "bedrock",
"family": "meta",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-8b-instruct-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "meta.llama4-maverick-17b-instruct-v1:0",
"name": "Llama 4 Maverick 17B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2025-04-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.24,
"output_per_million": 0.97
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-05",
"cost": {
"input": 0.24,
"output": 0.97
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-08"
}
},
{
"id": "meta.llama4-scout-17b-instruct-v1:0",
"name": "Llama 4 Scout 17B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2025-04-05 00:00:00 UTC",
"context_window": 3500000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.17,
"output_per_million": 0.66
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-05",
"cost": {
"input": 0.17,
"output": 0.66
},
"limit": {
"context": 3500000,
"output": 16384
},
"knowledge": "2024-08"
}
},
{
"id": "minimax.minimax-m2",
"name": "MiniMax M2",
"provider": "bedrock",
"family": "minimax",
"created_at": "2025-10-27 00:00:00 UTC",
"context_window": 204608,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2
}
}
},
"metadata": {
"provider_name": "MiniMax",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/minimax.minimax-m2",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 409600,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-10-27",
"cost": {
"input": 0.3,
"output": 1.2
},
"limit": {
"context": 204608,
"output": 128000
}
}
},
{
"id": "minimax.minimax-m2.1",
"name": "MiniMax M2.1",
"provider": "bedrock",
"family": "minimax",
"created_at": "2025-12-23 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2
}
}
},
"metadata": {
"provider_name": "MiniMax",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/minimax.minimax-m2.1",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 196608,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-23",
"cost": {
"input": 0.3,
"output": 1.2
},
"limit": {
"context": 204800,
"output": 131072
}
}
},
{
"id": "minimax.minimax-m2.5",
"name": "MiniMax M2.5",
"provider": "bedrock",
"family": "minimax",
"created_at": "2026-03-18 00:00:00 UTC",
"context_window": 196608,
"max_output_tokens": 98304,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2
}
}
},
"metadata": {
"provider_name": "MiniMax",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/minimax.minimax-m2.5",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 196608,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-18",
"cost": {
"input": 0.3,
"output": 1.2
},
"limit": {
"context": 196608,
"output": 98304
}
}
},
{
"id": "mistral.devstral-2-123b",
"name": "Devstral 2 123B",
"provider": "bedrock",
"family": "devstral",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.devstral-2-123b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-17",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 256000,
"output": 8192
}
}
},
{
"id": "mistral.magistral-small-2509",
"name": "Magistral Small 1.2",
"provider": "bedrock",
"family": "magistral",
"created_at": "2025-12-02 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 40000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.magistral-small-2509",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-02",
"cost": {
"input": 0.5,
"output": 1.5
},
"limit": {
"context": 128000,
"output": 40000
}
}
},
{
"id": "mistral.ministral-3-14b-instruct",
"name": "Ministral 14B 3.0",
"provider": "bedrock",
"family": "ministral",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.2
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.ministral-3-14b-instruct",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.2,
"output": 0.2
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "mistral.ministral-3-3b-instruct",
"name": "Ministral 3 3B",
"provider": "bedrock",
"family": "ministral",
"created_at": "2025-12-02 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.ministral-3-3b-instruct",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-02",
"cost": {
"input": 0.1,
"output": 0.1
},
"limit": {
"context": 256000,
"output": 8192
}
}
},
{
"id": "mistral.ministral-3-8b-instruct",
"name": "Ministral 3 8B",
"provider": "bedrock",
"family": "ministral",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.ministral-3-8b-instruct",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.15,
"output": 0.15
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "mistral.mistral-7b-instruct-v0:2",
"name": "Mistral 7B Instruct",
"provider": "bedrock",
"family": "mistral ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.mistral-7b-instruct-v0:2",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "mistral.mistral-large-2402-v1:0",
"name": "Mistral Large (24.02)",
"provider": "bedrock",
"family": "mistral ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.mistral-large-2402-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "mistral.mistral-large-2407-v1:0",
"name": "Mistral Large (24.07)",
"provider": "bedrock",
"family": "mistral ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.mistral-large-2407-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "mistral.mistral-large-3-675b-instruct",
"name": "Mistral Large 3",
"provider": "bedrock",
"family": "mistral",
"created_at": "2025-12-02 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.mistral-large-3-675b-instruct",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-02",
"cost": {
"input": 0.5,
"output": 1.5
},
"limit": {
"context": 256000,
"output": 8192
}
}
},
{
"id": "mistral.mixtral-8x7b-instruct-v0:1",
"name": "Mixtral 8x7B Instruct",
"provider": "bedrock",
"family": "mistral ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.mixtral-8x7b-instruct-v0:1",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "mistral.pixtral-large-2502-v1:0",
"name": "Pixtral Large (25.02)",
"provider": "bedrock",
"family": "mistral",
"created_at": "2025-04-08 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-04-08",
"cost": {
"input": 2,
"output": 6
},
"limit": {
"context": 128000,
"output": 8192
}
}
},
{
"id": "mistral.voxtral-mini-3b-2507",
"name": "Voxtral Mini 3B 2507",
"provider": "bedrock",
"family": "mistral",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"audio",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.04
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.voxtral-mini-3b-2507",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 32768,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.04,
"output": 0.04
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "mistral.voxtral-small-24b-2507",
"name": "Voxtral Small 24B 2507",
"provider": "bedrock",
"family": "mistral",
"created_at": "2025-07-01 00:00:00 UTC",
"context_window": 32000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"audio"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.35
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.voxtral-small-24b-2507",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 32768,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-07-01",
"cost": {
"input": 0.15,
"output": 0.35
},
"limit": {
"context": 32000,
"output": 8192
}
}
},
{
"id": "moonshot.kimi-k2-thinking",
"name": "Kimi K2 Thinking",
"provider": "bedrock",
"family": "kimi-thinking",
"created_at": "2025-12-02 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.5
}
}
},
"metadata": {
"provider_name": "Moonshot AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/moonshot.kimi-k2-thinking",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-02",
"interleaved": true,
"cost": {
"input": 0.6,
"output": 2.5
},
"limit": {
"context": 256000,
"output": 256000
}
}
},
{
"id": "moonshotai.kimi-k2.5",
"name": "Kimi K2.5",
"provider": "bedrock",
"family": "kimi",
"created_at": "2026-02-06 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 3
}
}
},
"metadata": {
"provider_name": "Moonshot AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/moonshotai.kimi-k2.5",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-06",
"interleaved": true,
"cost": {
"input": 0.6,
"output": 3
},
"limit": {
"context": 256000,
"output": 256000
}
}
},
{
"id": "nvidia.nemotron-nano-12b-v2",
"name": "NVIDIA Nemotron Nano 12B v2 VL BF16",
"provider": "bedrock",
"family": "nemotron",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.6
}
}
},
"metadata": {
"provider_name": "NVIDIA",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/nvidia.nemotron-nano-12b-v2",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.2,
"output": 0.6
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "nvidia.nemotron-nano-3-30b",
"name": "NVIDIA Nemotron Nano 3 30B",
"provider": "bedrock",
"family": "nemotron",
"created_at": "2025-12-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.24
}
}
},
"metadata": {
"provider_name": "NVIDIA",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/nvidia.nemotron-nano-3-30b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-23",
"cost": {
"input": 0.06,
"output": 0.24
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "nvidia.nemotron-nano-9b-v2",
"name": "NVIDIA Nemotron Nano 9B v2",
"provider": "bedrock",
"family": "nemotron",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.23
}
}
},
"metadata": {
"provider_name": "NVIDIA",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/nvidia.nemotron-nano-9b-v2",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.06,
"output": 0.23
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "nvidia.nemotron-super-3-120b",
"name": "NVIDIA Nemotron 3 Super 120B A12B",
"provider": "bedrock",
"family": "nemotron",
"created_at": "2026-03-11 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.65
}
}
},
"metadata": {
"provider_name": "NVIDIA",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/nvidia.nemotron-super-3-120b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-11",
"cost": {
"input": 0.15,
"output": 0.65
},
"limit": {
"context": 262144,
"output": 131072
}
}
},
{
"id": "openai.gpt-oss-120b-1:0",
"name": "gpt-oss-120b",
"provider": "bedrock",
"family": "gpt-oss",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"provider_name": "OpenAI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/openai.gpt-oss-120b-1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": null,
"maxTokensDefault": 4096,
"maxTokensMaximum": 128000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "openai.gpt-oss-20b-1:0",
"name": "gpt-oss-20b",
"provider": "bedrock",
"family": "gpt-oss",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.3
}
}
},
"metadata": {
"provider_name": "OpenAI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/openai.gpt-oss-20b-1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": null,
"maxTokensDefault": 4096,
"maxTokensMaximum": 128000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.07,
"output": 0.3
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "openai.gpt-oss-safeguard-120b",
"name": "GPT OSS Safeguard 120B",
"provider": "bedrock",
"family": "gpt-oss",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"provider_name": "OpenAI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/openai.gpt-oss-safeguard-120b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "openai.gpt-oss-safeguard-20b",
"name": "GPT OSS Safeguard 20B",
"provider": "bedrock",
"family": "gpt-oss",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.2
}
}
},
"metadata": {
"provider_name": "OpenAI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/openai.gpt-oss-safeguard-20b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.07,
"output": 0.2
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "qwen.qwen3-235b-a22b-2507-v1:0",
"name": "Qwen3 235B A22B 2507",
"provider": "bedrock",
"family": "qwen",
"created_at": "2025-09-18 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.22,
"output_per_million": 0.88
}
}
},
"metadata": {
"provider_name": "Qwen",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/qwen.qwen3-235b-a22b-2507-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-18",
"cost": {
"input": 0.22,
"output": 0.88
},
"limit": {
"context": 262144,
"output": 131072
},
"knowledge": "2024-04"
}
},
{
"id": "qwen.qwen3-32b-v1:0",
"name": "Qwen3 32B (dense)",
"provider": "bedrock",
"family": "qwen",
"created_at": "2025-09-18 00:00:00 UTC",
"context_window": 16384,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"provider_name": "Qwen",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/qwen.qwen3-32b-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":true}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 32768,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-18",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 16384,
"output": 16384
},
"knowledge": "2024-04"
}
},
{
"id": "qwen.qwen3-coder-30b-a3b-v1:0",
"name": "Qwen3 Coder 30B A3B Instruct",
"provider": "bedrock",
"family": "qwen",
"created_at": "2025-09-18 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"provider_name": "Qwen",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/qwen.qwen3-coder-30b-a3b-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-18",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 262144,
"output": 131072
},
"knowledge": "2024-04"
}
},
{
"id": "qwen.qwen3-coder-480b-a35b-v1:0",
"name": "Qwen3 Coder 480B A35B Instruct",
"provider": "bedrock",
"family": "qwen",
"created_at": "2025-09-18 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.22,
"output_per_million": 1.8
}
}
},
"metadata": {
"provider_name": "Qwen",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/qwen.qwen3-coder-480b-a35b-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "",
"maxTokensDefault": null,
"maxTokensMaximum": 131072,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-18",
"cost": {
"input": 0.22,
"output": 1.8
},
"limit": {
"context": 131072,
"output": 65536
},
"knowledge": "2024-04"
}
},
{
"id": "qwen.qwen3-coder-next",
"name": "Qwen3 Coder Next",
"provider": "bedrock",
"family": "qwen",
"created_at": "2026-02-06 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.22,
"output_per_million": 1.8
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-06",
"cost": {
"input": 0.22,
"output": 1.8
},
"limit": {
"context": 131072,
"output": 65536
}
}
},
{
"id": "qwen.qwen3-next-80b-a3b",
"name": "Qwen/Qwen3-Next-80B-A3B-Instruct",
"provider": "bedrock",
"family": "qwen",
"created_at": "2025-09-18 00:00:00 UTC",
"context_window": 262000,
"max_output_tokens": 262000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 1.4
}
}
},
"metadata": {
"provider_name": "Qwen",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/qwen.qwen3-next-80b-a3b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-11-25",
"cost": {
"input": 0.14,
"output": 1.4
},
"limit": {
"context": 262000,
"output": 262000
}
}
},
{
"id": "qwen.qwen3-vl-235b-a22b",
"name": "Qwen/Qwen3-VL-235B-A22B-Instruct",
"provider": "bedrock",
"family": "qwen",
"created_at": "2025-10-04 00:00:00 UTC",
"context_window": 262000,
"max_output_tokens": 262000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.5
}
}
},
"metadata": {
"provider_name": "Qwen",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/qwen.qwen3-vl-235b-a22b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 262144,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"png",
"jpeg",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-25",
"cost": {
"input": 0.3,
"output": 1.5
},
"limit": {
"context": 262000,
"output": 262000
}
}
},
{
"id": "stability.sd3-5-large-v1:0",
"name": "Stable Diffusion 3.5 Large",
"provider": "bedrock",
"family": "stability ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.sd3-5-large-v1:0",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "stability.stable-image-core-v1:1",
"name": "Stable Image Core 1.0",
"provider": "bedrock",
"family": "stability ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-core-v1:1",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "stability.stable-image-ultra-v1:1",
"name": "Stable Image Ultra 1.0",
"provider": "bedrock",
"family": "stability ai",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-ultra-v1:1",
"inference_types": [
"ON_DEMAND"
],
"converse": {}
}
},
{
"id": "us.amazon.nova-2-lite-v1:0",
"name": "Nova 2 Lite",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.33,
"output_per_million": 2.75
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-2-lite-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"topK\":{\"type\":\"number\",\"default\":50,\"minimum\":1,\"maximum\":100},\"reasoningConfig\":{\"budgetTokens\":{\"type\":\"enum\",\"default\":\"low\",\"minimum\":10000,\"maximum\":64000,\"enum\":{\"low\":10000,\"medium\":40000,\"high\":64000},\"smartSyncEnabled\":true}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 65535,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf",
"docx"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": [
"mkv",
"mov",
"mp4",
"webm",
"flv",
"mpeg",
"mpg",
"wmv",
"three_gp"
]
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-01",
"cost": {
"input": 0.33,
"output": 2.75
},
"limit": {
"context": 128000,
"output": 4096
}
}
},
{
"id": "us.amazon.nova-lite-v1:0",
"name": "Nova Lite",
"provider": "bedrock",
"family": "nova-lite",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 300000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.24,
"cache_read_input_per_million": 0.015
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-lite-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 0.06,
"output": 0.24,
"cache_read": 0.015
},
"limit": {
"context": 300000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "us.amazon.nova-micro-v1:0",
"name": "Nova Micro",
"provider": "bedrock",
"family": "nova-micro",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.035,
"output_per_million": 0.14,
"cache_read_input_per_million": 0.00875
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-micro-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 0.035,
"output": 0.14,
"cache_read": 0.00875
},
"limit": {
"context": 128000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "us.amazon.nova-premier-v1:0",
"name": "Nova Premier",
"provider": "bedrock",
"family": "nova",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 12.5
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-premier-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 2.5,
"output": 12.5
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "us.amazon.nova-pro-v1:0",
"name": "Nova Pro",
"provider": "bedrock",
"family": "nova-pro",
"created_at": "2024-12-03 00:00:00 UTC",
"context_window": 300000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 3.2,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"provider_name": "Amazon",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/amazon.nova-pro-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": null,
"maxTokensDefault": null,
"maxTokensMaximum": 10000,
"reasoningSupported": null,
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf",
"docx"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": [
"mkv",
"mov",
"mp4",
"webm",
"flv",
"mpeg",
"mpg",
"wmv",
"three_gp"
]
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-03",
"cost": {
"input": 0.8,
"output": 3.2,
"cache_read": 0.2
},
"limit": {
"context": 300000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "us.anthropic.claude-haiku-4-5-20251001-v1:0",
"name": "Claude Haiku 4.5 (US)",
"provider": "bedrock",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-haiku-4-5-20251001-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"top_k\":{\"type\":\"number\",\"default\":250,\"minimum\":0,\"maximum\":500},\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"integer\",\"default\":2048,\"minimum\":1024,\"maximum\":63999}}}",
"maxTokensDefault": 64000,
"maxTokensMaximum": 64000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "us.anthropic.claude-opus-4-1-20250805-v1:0",
"name": "Claude Opus 4.1 (US)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-opus-4-1-20250805-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"topK\":{\"type\":\"number\",\"default\":250,\"minimum\":0,\"maximum\":500},\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"integer\",\"default\":2048,\"minimum\":1024,\"maximum\":63999}}}",
"maxTokensDefault": 1024,
"maxTokensMaximum": 32000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "us.anthropic.claude-opus-4-20250514-v1:0",
"name": "Claude Opus 4 (US)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-opus-4-20250514-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "us.anthropic.claude-opus-4-5-20251101-v1:0",
"name": "Claude Opus 4.5 (US)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2025-11-24 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-opus-4-5-20251101-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"top_k\":{\"type\":\"number\",\"default\":250,\"minimum\":0,\"maximum\":500},\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"integer\",\"default\":2048,\"minimum\":1024,\"maximum\":63999}}}",
"maxTokensDefault": 64000,
"maxTokensMaximum": 64000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-01",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "us.anthropic.claude-opus-4-6-v1",
"name": "Claude Opus 4.6 (US)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-opus-4-6-v1",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"top_k\":{\"type\":\"number\",\"default\":250,\"minimum\":0,\"maximum\":500},\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"enum\",\"default\":\"low\",\"enum\":{\"low\":1024,\"medium\":40000,\"high\":63999},\"minimum\":1024,\"maximum\":63999}}}",
"maxTokensDefault": 128000,
"maxTokensMaximum": 128000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "us.anthropic.claude-opus-4-7",
"name": "Claude Opus 4.7 (US)",
"provider": "bedrock",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-opus-4-7",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"enum\",\"default\":\"low\",\"enum\":{\"low\":1024,\"medium\":40000,\"high\":63999},\"minimum\":1024,\"maximum\":63999}},\"hideSamplingParameter\":true}",
"maxTokensDefault": 4096,
"maxTokensMaximum": 128000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "us.anthropic.claude-sonnet-4-20250514-v1:0",
"name": "Claude Sonnet 4 (US)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-sonnet-4-20250514-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": null,
"maxTokensDefault": 8192,
"maxTokensMaximum": 65536,
"reasoningSupported": {
"embedded": true
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "us.anthropic.claude-sonnet-4-5-20250929-v1:0",
"name": "Claude Sonnet 4.5 (US)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-sonnet-4-5-20250929-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"top_k\":{\"type\":\"number\",\"default\":250,\"minimum\":0,\"maximum\":500},\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"integer\",\"default\":2048,\"minimum\":1024,\"maximum\":63999}}}",
"maxTokensDefault": 64000,
"maxTokensMaximum": 64000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": null,
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "us.anthropic.claude-sonnet-4-6",
"name": "Claude Sonnet 4.6 (US)",
"provider": "bedrock",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"provider_name": "Anthropic",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/anthropic.claude-sonnet-4-6",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {
"additionalRequestFieldsSchema": "{\"top_k\":{\"type\":\"number\",\"default\":250,\"minimum\":0,\"maximum\":500},\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false},\"budgetTokens\":{\"type\":\"enum\",\"default\":\"low\",\"enum\":{\"low\":1024,\"medium\":40000,\"high\":63999},\"minimum\":1024,\"maximum\":63999}}}",
"maxTokensDefault": 128000,
"maxTokensMaximum": 128000,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [
"pdf"
],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-08-31"
}
},
{
"id": "us.cohere.embed-v4:0",
"name": "Embed v4",
"provider": "bedrock",
"family": "Embed",
"created_at": null,
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"embeddings"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Cohere",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/cohere.embed-v4:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.deepseek.r1-v1:0",
"name": "DeepSeek-R1",
"provider": "bedrock",
"family": "deepseek-thinking",
"created_at": "2025-01-20 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.35,
"output_per_million": 5.4
}
}
},
"metadata": {
"provider_name": "DeepSeek",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/deepseek.r1-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-05-29",
"cost": {
"input": 1.35,
"output": 5.4
},
"limit": {
"context": 128000,
"output": 32768
},
"knowledge": "2024-07"
}
},
{
"id": "us.meta.llama3-2-11b-instruct-v1:0",
"name": "Llama 3.2 11B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.16,
"output_per_million": 0.16
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-11b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.16,
"output": 0.16
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "us.meta.llama3-2-1b-instruct-v1:0",
"name": "Llama 3.2 1B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 131000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-1b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.1,
"output": 0.1
},
"limit": {
"context": 131000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "us.meta.llama3-2-3b-instruct-v1:0",
"name": "Llama 3.2 3B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 131000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-3b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.15,
"output": 0.15
},
"limit": {
"context": 131000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "us.meta.llama3-2-90b-instruct-v1:0",
"name": "Llama 3.2 90B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-2-90b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "us.meta.llama3-3-70b-instruct-v1:0",
"name": "Llama 3.3 70B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2024-12-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.72,
"output_per_million": 0.72
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama3-3-70b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-06",
"cost": {
"input": 0.72,
"output": 0.72
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "us.meta.llama4-maverick-17b-instruct-v1:0",
"name": "Llama 4 Maverick 17B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2025-04-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.24,
"output_per_million": 0.97
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama4-maverick-17b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-05",
"cost": {
"input": 0.24,
"output": 0.97
},
"limit": {
"context": 1000000,
"output": 16384
},
"knowledge": "2024-08"
}
},
{
"id": "us.meta.llama4-scout-17b-instruct-v1:0",
"name": "Llama 4 Scout 17B Instruct",
"provider": "bedrock",
"family": "llama",
"created_at": "2025-04-05 00:00:00 UTC",
"context_window": 3500000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.17,
"output_per_million": 0.66
}
}
},
"metadata": {
"provider_name": "Meta",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/meta.llama4-scout-17b-instruct-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-05",
"cost": {
"input": 0.17,
"output": 0.66
},
"limit": {
"context": 3500000,
"output": 16384
},
"knowledge": "2024-08"
}
},
{
"id": "us.mistral.pixtral-large-2502-v1:0",
"name": "Pixtral Large (25.02)",
"provider": "bedrock",
"family": "mistral",
"created_at": "2025-04-08 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6
}
}
},
"metadata": {
"provider_name": "Mistral AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/mistral.pixtral-large-2502-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-04-08",
"cost": {
"input": 2,
"output": 6
},
"limit": {
"context": 128000,
"output": 8192
}
}
},
{
"id": "us.stability.stable-conservative-upscale-v1:0",
"name": "Stable Image Conservative Upscale",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-conservative-upscale-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-creative-upscale-v1:0",
"name": "Stable Image Creative Upscale",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-creative-upscale-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-fast-upscale-v1:0",
"name": "Stable Image Fast Upscale",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-fast-upscale-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-control-sketch-v1:0",
"name": "Stable Image Control Sketch",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-control-sketch-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-control-structure-v1:0",
"name": "Stable Image Control Structure",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-control-structure-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-erase-object-v1:0",
"name": "Stable Image Erase Object",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-erase-object-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-inpaint-v1:0",
"name": "Stable Image Inpaint",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-inpaint-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-remove-background-v1:0",
"name": "Stable Image Remove Background",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-remove-background-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-search-recolor-v1:0",
"name": "Stable Image Search and Recolor",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-search-recolor-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-search-replace-v1:0",
"name": "Stable Image Search and Replace",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-search-replace-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-image-style-guide-v1:0",
"name": "Stable Image Style Guide",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-image-style-guide-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-outpaint-v1:0",
"name": "Stable Image Outpaint",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-outpaint-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.stability.stable-style-transfer-v1:0",
"name": "Stable Image Style Transfer",
"provider": "bedrock",
"family": "Stable Image Services",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Stability AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/stability.stable-style-transfer-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.twelvelabs.pegasus-1-2-v1:0",
"name": "Pegasus v1.2",
"provider": "bedrock",
"family": "Pegasus",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "TwelveLabs",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/twelvelabs.pegasus-1-2-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {}
}
},
{
"id": "us.writer.palmyra-x4-v1:0",
"name": "Palmyra X4",
"provider": "bedrock",
"family": "palmyra",
"created_at": "2025-04-28 00:00:00 UTC",
"context_window": 122880,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10
}
}
},
"metadata": {
"provider_name": "Writer",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/writer.palmyra-x4-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-04-28",
"cost": {
"input": 2.5,
"output": 10
},
"limit": {
"context": 122880,
"output": 8192
}
}
},
{
"id": "us.writer.palmyra-x5-v1:0",
"name": "Palmyra X5",
"provider": "bedrock",
"family": "palmyra",
"created_at": "2025-04-28 00:00:00 UTC",
"context_window": 1040000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 6
}
}
},
"metadata": {
"provider_name": "Writer",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/writer.palmyra-x5-v1:0",
"inference_types": [
"INFERENCE_PROFILE"
],
"converse": {},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-04-28",
"cost": {
"input": 0.6,
"output": 6
},
"limit": {
"context": 1040000,
"output": 8192
}
}
},
{
"id": "writer.palmyra-vision-7b",
"name": "Writer Palmyra Vision 7B",
"provider": "bedrock",
"family": "Writer Palmyra Vision",
"created_at": null,
"context_window": null,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"provider_name": "Writer",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/writer.palmyra-vision-7b",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{}",
"maxTokensDefault": null,
"maxTokensMaximum": 4096,
"reasoningSupported": null,
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [
"jpeg",
"png",
"gif",
"webp"
],
"userVideoTypesSupported": []
}
}
},
{
"id": "writer.palmyra-x4-v1:0",
"name": "Palmyra X4",
"provider": "bedrock",
"family": "palmyra",
"created_at": "2025-04-28 00:00:00 UTC",
"context_window": 122880,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-04-28",
"cost": {
"input": 2.5,
"output": 10
},
"limit": {
"context": 122880,
"output": 8192
}
}
},
{
"id": "writer.palmyra-x5-v1:0",
"name": "Palmyra X5",
"provider": "bedrock",
"family": "palmyra",
"created_at": "2025-04-28 00:00:00 UTC",
"context_window": 1040000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 6
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-04-28",
"cost": {
"input": 0.6,
"output": 6
},
"limit": {
"context": 1040000,
"output": 8192
}
}
},
{
"id": "zai.glm-4.7",
"name": "GLM-4.7",
"provider": "bedrock",
"family": "glm",
"created_at": "2025-12-22 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.2
}
}
},
"metadata": {
"provider_name": "Z.AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/zai.glm-4.7",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 202752,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-22",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0.6,
"output": 2.2
},
"limit": {
"context": 204800,
"output": 131072
},
"knowledge": "2025-04"
}
},
{
"id": "zai.glm-4.7-flash",
"name": "GLM-4.7-Flash",
"provider": "bedrock",
"family": "glm-flash",
"created_at": "2026-01-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.4
}
}
},
"metadata": {
"provider_name": "Z.AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/zai.glm-4.7-flash",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 202752,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-19",
"cost": {
"input": 0.07,
"output": 0.4
},
"limit": {
"context": 200000,
"output": 131072
},
"knowledge": "2025-04"
}
},
{
"id": "zai.glm-5",
"name": "GLM-5",
"provider": "bedrock",
"family": "glm",
"created_at": "2026-03-18 00:00:00 UTC",
"context_window": 202752,
"max_output_tokens": 101376,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 3.2
}
}
},
"metadata": {
"provider_name": "Z.AI",
"model_arn": "arn:aws:bedrock:us-west-2::foundation-model/zai.glm-5",
"inference_types": [
"ON_DEMAND"
],
"converse": {
"additionalRequestFieldsSchema": "{\"reasoningConfig\":{\"enabled\":{\"type\":\"boolean\",\"default\":false}}}",
"maxTokensDefault": null,
"maxTokensMaximum": 202752,
"reasoningSupported": {
"embedded": false
},
"stopSequencesDefault": [],
"systemRoleSupported": true,
"userDocumentTypesSupported": [],
"userImageTypesSupported": [],
"userVideoTypesSupported": []
},
"source": "models.dev",
"provider_id": "amazon-bedrock",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-18",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 1,
"output": 3.2
},
"limit": {
"context": 202752,
"output": 101376
}
}
},
{
"id": "deepseek-chat",
"name": "DeepSeek Chat",
"provider": "deepseek",
"family": "deepseek",
"created_at": "2025-12-01 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 384000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.28,
"cache_read_input_per_million": 0.028
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "deepseek",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-28",
"cost": {
"input": 0.14,
"output": 0.28,
"cache_read": 0.028
},
"limit": {
"context": 1000000,
"output": 384000
},
"knowledge": "2025-09"
}
},
{
"id": "deepseek-reasoner",
"name": "DeepSeek Reasoner",
"provider": "deepseek",
"family": "deepseek-thinking",
"created_at": "2025-12-01 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 384000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.28,
"cache_read_input_per_million": 0.028
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "deepseek",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-28",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0.14,
"output": 0.28,
"cache_read": 0.028
},
"limit": {
"context": 1000000,
"output": 384000
},
"knowledge": "2025-09"
}
},
{
"id": "deepseek-v4-flash",
"name": "DeepSeek V4 Flash",
"provider": "deepseek",
"family": "deepseek-flash",
"created_at": "2026-04-24 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 384000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.28,
"cache_read_input_per_million": 0.028
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "deepseek",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-24",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0.14,
"output": 0.28,
"cache_read": 0.028
},
"limit": {
"context": 1000000,
"output": 384000
},
"knowledge": "2025-05"
}
},
{
"id": "deepseek-v4-pro",
"name": "DeepSeek V4 Pro",
"provider": "deepseek",
"family": "deepseek-thinking",
"created_at": "2026-04-24 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 384000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.74,
"output_per_million": 3.48,
"cache_read_input_per_million": 0.145
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "deepseek",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-24",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 1.74,
"output": 3.48,
"cache_read": 0.145
},
"limit": {
"context": 1000000,
"output": 384000
},
"knowledge": "2025-05"
}
},
{
"id": "aqa",
"name": "Model that performs Attributed Question Answering.",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 7168,
"max_output_tokens": 1024,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {},
"metadata": {
"version": "001",
"description": "Model trained to return answers to questions that are grounded in provided sources, along with estimating answerable probability.",
"supported_generation_methods": [
"generateAnswer"
]
}
},
{
"id": "deep-research-max-preview-04-2026",
"name": "Deep Research Max Preview (Apr-21-2026)",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "deepthink-exp-05-20",
"description": "Preview release (April 21st, 2026) of Deep Research Max",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "deep-research-preview-04-2026",
"name": "Deep Research Preview (Apr-21-2026)",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "deepthink-exp-05-20",
"description": "Preview release (April 21th, 2026) of Deep Research",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "deep-research-pro-preview-12-2025",
"name": "Deep Research Pro Preview (Dec-12-2025)",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "deepthink-exp-05-20",
"description": "Preview release (December 12th, 2025) of Deep Research Pro",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "gemini-1.5-flash",
"name": "Gemini 1.5 Flash",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2024-05-14 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3,
"cache_read_input_per_million": 0.01875
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-05-14",
"cost": {
"input": 0.075,
"output": 0.3,
"cache_read": 0.01875
},
"limit": {
"context": 1000000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "gemini-1.5-flash-8b",
"name": "Gemini 1.5 Flash-8B",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2024-10-03 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.0375,
"output_per_million": 0.15,
"cache_read_input_per_million": 0.01
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-03",
"cost": {
"input": 0.0375,
"output": 0.15,
"cache_read": 0.01
},
"limit": {
"context": 1000000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "gemini-1.5-pro",
"name": "Gemini 1.5 Pro",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2024-02-15 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5,
"cache_read_input_per_million": 0.3125
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-02-15",
"cost": {
"input": 1.25,
"output": 5,
"cache_read": 0.3125
},
"limit": {
"context": 1000000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "gemini-2.0-flash",
"name": "Gemini 2.0 Flash",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2024-12-11 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"version": "2.0",
"description": "Gemini 2.0 Flash",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-11",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 8192
},
"knowledge": "2024-06"
}
},
{
"id": "gemini-2.0-flash-001",
"name": "Gemini 2.0 Flash 001",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4
}
}
},
"metadata": {
"version": "2.0",
"description": "Stable version of Gemini 2.0 Flash, our fast and versatile multimodal model for scaling across diverse tasks, released in January of 2025.",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
]
}
},
{
"id": "gemini-2.0-flash-lite",
"name": "Gemini 2.0 Flash Lite",
"provider": "gemini",
"family": "gemini-flash-lite",
"created_at": "2024-12-11 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "2.0",
"description": "Gemini 2.0 Flash-Lite",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-11",
"cost": {
"input": 0.075,
"output": 0.3
},
"limit": {
"context": 1048576,
"output": 8192
},
"knowledge": "2024-06"
}
},
{
"id": "gemini-2.0-flash-lite-001",
"name": "Gemini 2.0 Flash-Lite 001",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "2.0",
"description": "Stable version of Gemini 2.0 Flash-Lite",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
]
}
},
{
"id": "gemini-2.5-computer-use-preview-10-2025",
"name": "Gemini 2.5 Computer Use Preview 10-2025",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "Gemini 2.5 Computer Use Preview 10-2025",
"description": "Gemini 2.5 Computer Use Preview 10-2025",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "gemini-2.5-flash",
"name": "Gemini 2.5 Flash",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-03-20 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.03
}
},
"audio_tokens": {
"standard": {
"input_per_million": 1
}
}
},
"metadata": {
"version": "001",
"description": "Stable version of Gemini 2.5 Flash, our mid-size multimodal model that supports up to 1 million tokens, released in June of 2025.",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.03,
"input_audio": 1
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-image",
"name": "Gemini 2.5 Flash Image",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-08-26 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"reasoning",
"vision",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 30,
"cache_read_input_per_million": 0.075
}
}
},
"metadata": {
"version": "2.0",
"description": "Gemini 2.5 Flash Preview Image",
"supported_generation_methods": [
"generateContent",
"countTokens",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-26",
"cost": {
"input": 0.3,
"output": 30,
"cache_read": 0.075
},
"limit": {
"context": 32768,
"output": 32768
},
"knowledge": "2025-06"
}
},
{
"id": "gemini-2.5-flash-image-preview",
"name": "Gemini 2.5 Flash Image (Preview)",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-08-26 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 30,
"cache_read_input_per_million": 0.075
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-26",
"cost": {
"input": 0.3,
"output": 30,
"cache_read": 0.075
},
"limit": {
"context": 32768,
"output": 32768
},
"knowledge": "2025-06"
}
},
{
"id": "gemini-2.5-flash-lite",
"name": "Gemini 2.5 Flash Lite",
"provider": "gemini",
"family": "gemini-flash-lite",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"version": "001",
"description": "Stable version of Gemini 2.5 Flash-Lite, released in July of 2025",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-lite-preview-06-17",
"name": "Gemini 2.5 Flash Lite Preview 06-17",
"provider": "gemini",
"family": "gemini-flash-lite",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
},
"audio_tokens": {
"standard": {
"input_per_million": 0.3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025,
"input_audio": 0.3
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-lite-preview-09-2025",
"name": "Gemini 2.5 Flash Lite Preview 09-25",
"provider": "gemini",
"family": "gemini-flash-lite",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-native-audio-latest",
"name": "Gemini 2.5 Flash Native Audio Latest",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "Gemini 2.5 Flash Native Audio Latest",
"description": "Latest release of Gemini 2.5 Flash Native Audio",
"supported_generation_methods": [
"countTokens",
"bidiGenerateContent"
]
}
},
{
"id": "gemini-2.5-flash-native-audio-preview-09-2025",
"name": "Gemini 2.5 Flash Native Audio Preview 09-2025",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "gemini-2.5-flash-preview-native-audio-dialog-2025-05-19",
"description": "Gemini 2.5 Flash Native Audio Preview 09-2025",
"supported_generation_methods": [
"countTokens",
"bidiGenerateContent"
]
}
},
{
"id": "gemini-2.5-flash-native-audio-preview-12-2025",
"name": "Gemini 2.5 Flash Native Audio Preview 12-2025",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "12-2025",
"description": "Gemini 2.5 Flash Native Audio Preview 12-2025",
"supported_generation_methods": [
"countTokens",
"bidiGenerateContent"
]
}
},
{
"id": "gemini-2.5-flash-preview-04-17",
"name": "Gemini 2.5 Flash Preview 04-17",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-04-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.0375
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-17",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.0375
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-preview-05-20",
"name": "Gemini 2.5 Flash Preview 05-20",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.0375
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-20",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.0375
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-preview-09-2025",
"name": "Gemini 2.5 Flash Preview 09-25",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.075
}
},
"audio_tokens": {
"standard": {
"input_per_million": 1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.075,
"input_audio": 1
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-preview-tts",
"name": "Gemini 2.5 Flash Preview TTS",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-05-01 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 16000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"audio"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 10
}
}
},
"metadata": {
"version": "gemini-2.5-flash-exp-tts-2025-05-19",
"description": "Gemini 2.5 Flash Preview TTS",
"supported_generation_methods": [
"countTokens",
"generateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-05-01",
"cost": {
"input": 0.5,
"output": 10
},
"limit": {
"context": 8000,
"output": 16000
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro",
"name": "Gemini 2.5 Pro",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2025-03-20 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"version": "2.5",
"description": "Stable release (June 17th, 2025) of Gemini 2.5 Pro",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125,
"context_over_200k": {
"input": 2.5,
"output": 15,
"cache_read": 0.25
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro-preview-05-06",
"name": "Gemini 2.5 Pro Preview 05-06",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2025-05-06 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.31
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-06",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.31
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro-preview-06-05",
"name": "Gemini 2.5 Pro Preview 06-05",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2025-06-05 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.31
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.31
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro-preview-tts",
"name": "Gemini 2.5 Pro Preview TTS",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-05-01 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 16000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"audio"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 20
}
}
},
"metadata": {
"version": "gemini-2.5-pro-preview-tts-2025-05-19",
"description": "Gemini 2.5 Pro Preview TTS",
"supported_generation_methods": [
"countTokens",
"generateContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-05-01",
"cost": {
"input": 1,
"output": 20
},
"limit": {
"context": 8000,
"output": 16000
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3-flash-preview",
"name": "Gemini 3 Flash Preview",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-12-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 3,
"cache_read_input_per_million": 0.05
}
}
},
"metadata": {
"version": "3-flash-preview-12-2025",
"description": "Gemini 3 Flash Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-12-17",
"cost": {
"input": 0.5,
"output": 3,
"cache_read": 0.05,
"context_over_200k": {
"input": 0.5,
"output": 3,
"cache_read": 0.05
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3-pro-image-preview",
"name": "Nano Banana Pro",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.0",
"description": "Gemini 3 Pro Image Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"batchGenerateContent"
]
}
},
{
"id": "gemini-3-pro-preview",
"name": "Gemini 3 Pro Preview",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2025-11-18 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"version": "3-pro-preview-11-2025",
"description": "Gemini 3 Pro Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-18",
"cost": {
"input": 2,
"output": 12,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-flash-image-preview",
"name": "Gemini 3.1 Flash Image (Preview)",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2026-02-26 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"reasoning",
"vision",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 60
}
}
},
"metadata": {
"version": "3.0",
"description": "Gemini 3.1 Flash Image Preview.",
"supported_generation_methods": [
"generateContent",
"countTokens",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-26",
"cost": {
"input": 0.25,
"output": 60
},
"limit": {
"context": 131072,
"output": 32768
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-flash-lite-preview",
"name": "Gemini 3.1 Flash Lite Preview",
"provider": "gemini",
"family": "gemini-flash-lite",
"created_at": "2026-03-03 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.5,
"cache_read_input_per_million": 0.025,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"version": "3.1-flash-lite-preview-03-2026",
"description": "Gemini 3.1 Flash Lite Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-03",
"cost": {
"input": 0.25,
"output": 1.5,
"cache_read": 0.025,
"cache_write": 1
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-flash-live-preview",
"name": "Gemini 3.1 Flash Live Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.1-flash-live-03-2026",
"description": "Gemini 3.1 Flash Live Preview",
"supported_generation_methods": [
"bidiGenerateContent"
]
}
},
{
"id": "gemini-3.1-flash-tts-preview",
"name": "Gemini 3.1 Flash TTS Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 8192,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.1-flash-tts-preview",
"description": "Gemini 3.1 Flash TTS Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"batchGenerateContent"
]
}
},
{
"id": "gemini-3.1-pro-preview",
"name": "Gemini 3.1 Pro Preview",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2026-02-19 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"version": "3.1-pro-preview-01-2026",
"description": "Gemini 3.1 Pro Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-19",
"cost": {
"input": 2,
"output": 12,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-pro-preview-customtools",
"name": "Gemini 3.1 Pro Preview Custom Tools",
"provider": "gemini",
"family": "gemini-pro",
"created_at": "2026-02-19 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"version": "3.1-pro-preview-01-2026",
"description": "Gemini 3.1 Pro Preview optimized for custom tool usage",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-19",
"cost": {
"input": 2,
"output": 12,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-embedding-001",
"name": "Gemini Embedding 001",
"provider": "gemini",
"family": "gemini",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 2048,
"max_output_tokens": 3072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15
}
}
},
"metadata": {
"version": "001",
"description": "Obtain a distributed representation of a text.",
"supported_generation_methods": [
"embedContent",
"countTextTokens",
"countTokens",
"asyncBatchEmbedContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-05-20",
"cost": {
"input": 0.15,
"output": 0
},
"limit": {
"context": 2048,
"output": 3072
},
"knowledge": "2025-05"
}
},
{
"id": "gemini-embedding-2",
"name": "Gemini Embedding 2",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 8192,
"max_output_tokens": 1,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {},
"metadata": {
"version": "2",
"description": "Obtain a distributed representation of multimodal content.",
"supported_generation_methods": [
"embedContent",
"countTextTokens",
"countTokens",
"asyncBatchEmbedContent"
]
}
},
{
"id": "gemini-embedding-2-preview",
"name": "Gemini Embedding 2 Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 8192,
"max_output_tokens": 1,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {},
"metadata": {
"version": "2",
"description": "Obtain a distributed representation of multimodal content.",
"supported_generation_methods": [
"embedContent",
"countTextTokens",
"countTokens",
"asyncBatchEmbedContent"
]
}
},
{
"id": "gemini-flash-latest",
"name": "Gemini Flash Latest",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.075
}
},
"audio_tokens": {
"standard": {
"input_per_million": 1
}
}
},
"metadata": {
"version": "Gemini Flash Latest",
"description": "Latest release of Gemini Flash",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.075,
"input_audio": 1
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-flash-lite-latest",
"name": "Gemini Flash-Lite Latest",
"provider": "gemini",
"family": "gemini-flash-lite",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"version": "Gemini Flash-Lite Latest",
"description": "Latest release of Gemini Flash-Lite",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-live-2.5-flash",
"name": "Gemini Live 2.5 Flash",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-09-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 2
}
},
"audio_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 12
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-01",
"cost": {
"input": 0.5,
"output": 2,
"input_audio": 3,
"output_audio": 12
},
"limit": {
"context": 128000,
"output": 8000
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-live-2.5-flash-preview-native-audio",
"name": "Gemini Live 2.5 Flash Preview Native Audio",
"provider": "gemini",
"family": "gemini-flash",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"audio",
"video"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 2
}
},
"audio_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 12
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-09-18",
"cost": {
"input": 0.5,
"output": 2,
"input_audio": 3,
"output_audio": 12
},
"limit": {
"context": 131072,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-pro-latest",
"name": "Gemini Pro Latest",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "Gemini Pro Latest",
"description": "Latest release of Gemini Pro",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
]
}
},
{
"id": "gemini-robotics-er-1.5-preview",
"name": "Gemini Robotics-ER 1.5 Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "1.5-preview",
"description": "Gemini Robotics-ER 1.5 Preview",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "gemini-robotics-er-1.6-preview",
"name": "Gemini Robotics-ER 1.6 Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "1.6-preview",
"description": "Gemini Robotics-ER 1.6 Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"createCachedContent",
"batchGenerateContent"
]
}
},
{
"id": "gemma-3-12b-it",
"name": "Gemma 3 12B",
"provider": "gemini",
"family": "gemma",
"created_at": "2025-03-13 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-13",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "gemma-3-27b-it",
"name": "Gemma 3 27B",
"provider": "gemini",
"family": "gemma",
"created_at": "2025-03-12 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-12",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "gemma-3-4b-it",
"name": "Gemma 3 4B",
"provider": "gemini",
"family": "gemma",
"created_at": "2025-03-13 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-13",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "gemma-3n-e2b-it",
"name": "Gemma 3n 2B",
"provider": "gemini",
"family": "gemma",
"created_at": "2025-07-09 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-07-09",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 2000
},
"knowledge": "2024-10"
}
},
{
"id": "gemma-3n-e4b-it",
"name": "Gemma 3n 4B",
"provider": "gemini",
"family": "gemma",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-20",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 2000
},
"knowledge": "2024-10"
}
},
{
"id": "gemma-4-26b-a4b-it",
"name": "Gemma 4 26B",
"provider": "gemini",
"family": "gemma",
"created_at": "2026-04-02 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "001",
"description": "Gemma 4 26B A4B IT",
"supported_generation_methods": [
"generateContent",
"countTokens"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-02",
"limit": {
"context": 256000,
"output": 8192
}
}
},
{
"id": "gemma-4-31b-it",
"name": "Gemma 4 31B",
"provider": "gemini",
"family": "gemma",
"created_at": "2026-04-02 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "001",
"description": "Gemma 4 31B IT",
"supported_generation_methods": [
"generateContent",
"countTokens"
],
"source": "models.dev",
"provider_id": "google",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-02",
"limit": {
"context": 256000,
"output": 8192
}
}
},
{
"id": "imagen-4.0-fast-generate-001",
"name": "Imagen 4 Fast",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.03
}
}
},
"metadata": {
"version": "001",
"description": "Vertex served Imagen 4.0 Fast model",
"supported_generation_methods": [
"predict"
]
}
},
{
"id": "imagen-4.0-generate-001",
"name": "Imagen 4",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.03
}
}
},
"metadata": {
"version": "001",
"description": "Vertex served Imagen 4.0 model",
"supported_generation_methods": [
"predict"
]
}
},
{
"id": "imagen-4.0-ultra-generate-001",
"name": "Imagen 4 Ultra",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.03
}
}
},
"metadata": {
"version": "001",
"description": "Vertex served Imagen 4.0 ultra model",
"supported_generation_methods": [
"predict"
]
}
},
{
"id": "lyria-3-clip-preview",
"name": "Lyria 3 Clip Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "lyria-3-clip-preview",
"description": "Lyria 3 30s model Preview",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "lyria-3-pro-preview",
"name": "Lyria 3 Pro Preview",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "lyria-3-pro-preview",
"description": "Lyria 3 Pro Preview",
"supported_generation_methods": [
"generateContent",
"countTokens"
]
}
},
{
"id": "nano-banana-pro-preview",
"name": "Nano Banana Pro",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.0",
"description": "Gemini 3 Pro Image Preview",
"supported_generation_methods": [
"generateContent",
"countTokens",
"batchGenerateContent"
]
}
},
{
"id": "veo-2.0-generate-001",
"name": "Veo 2",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "2.0",
"description": "Vertex served Veo 2 model. Access to this model requires billing to be enabled on the associated Google Cloud Platform account. Please visit https://console.cloud.google.com/billing to enable it.",
"supported_generation_methods": [
"predictLongRunning"
]
}
},
{
"id": "veo-3.0-fast-generate-001",
"name": "Veo 3 fast",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.0",
"description": "Veo 3 fast",
"supported_generation_methods": [
"predictLongRunning"
]
}
},
{
"id": "veo-3.0-generate-001",
"name": "Veo 3",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.0",
"description": "Veo 3",
"supported_generation_methods": [
"predictLongRunning"
]
}
},
{
"id": "veo-3.1-fast-generate-preview",
"name": "Veo 3.1 fast",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.1",
"description": "Veo 3.1 fast",
"supported_generation_methods": [
"predictLongRunning"
]
}
},
{
"id": "veo-3.1-generate-preview",
"name": "Veo 3.1",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.1",
"description": "Veo 3.1",
"supported_generation_methods": [
"predictLongRunning"
]
}
},
{
"id": "veo-3.1-lite-generate-preview",
"name": "Veo 3.1 lite",
"provider": "gemini",
"family": null,
"created_at": null,
"context_window": 480,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"version": "3.1",
"description": "Veo 3.1 lite",
"supported_generation_methods": [
"predictLongRunning"
]
}
},
{
"id": "codestral-2508",
"name": "Codestral",
"provider": "mistral",
"family": "codestral",
"created_at": "2025-08-29 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch",
"predicted_outputs"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "codestral-embed",
"name": "Codestral",
"provider": "mistral",
"family": "codestral",
"created_at": "2025-05-20 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"predicted_outputs"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "codestral-embed-2505",
"name": "Codestral",
"provider": "mistral",
"family": "codestral",
"created_at": "2025-05-20 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"predicted_outputs"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "codestral-latest",
"name": "Codestral (latest)",
"provider": "mistral",
"family": "codestral",
"created_at": "2024-05-29 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.9
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-01-04",
"cost": {
"input": 0.3,
"output": 0.9
},
"limit": {
"context": 256000,
"output": 4096
},
"knowledge": "2024-10"
}
},
{
"id": "devstral-2512",
"name": "Devstral 2",
"provider": "mistral",
"family": "devstral",
"created_at": "2025-12-09 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-09",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-12"
}
},
{
"id": "devstral-latest",
"name": "Devstral Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "devstral-medium-2507",
"name": "Devstral Medium",
"provider": "mistral",
"family": "devstral",
"created_at": "2025-07-10 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-10",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2025-05"
}
},
{
"id": "devstral-medium-latest",
"name": "Devstral 2 (latest)",
"provider": "mistral",
"family": "devstral",
"created_at": "2025-12-02 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-02",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-12"
}
},
{
"id": "devstral-small-2505",
"name": "Devstral Small 2505",
"provider": "mistral",
"family": "devstral",
"created_at": "2025-05-07 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-05-07",
"cost": {
"input": 0.1,
"output": 0.3
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2025-05"
}
},
{
"id": "devstral-small-2507",
"name": "Devstral Small",
"provider": "mistral",
"family": "devstral",
"created_at": "2025-07-10 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.3
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-10",
"cost": {
"input": 0.1,
"output": 0.3
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2025-05"
}
},
{
"id": "labs-devstral-small-2512",
"name": "Devstral Small 2",
"provider": "mistral",
"family": "devstral",
"created_at": "2025-12-09 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-09",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 256000,
"output": 256000
},
"knowledge": "2025-12"
}
},
{
"id": "labs-leanstral-2603",
"name": "Labs Leanstral 2603",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "magistral-medium-2509",
"name": "Magistral Medium 2509",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "magistral-medium-latest",
"name": "Magistral Medium (latest)",
"provider": "mistral",
"family": "magistral-medium",
"created_at": "2025-03-17 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output",
"batch"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 5
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-03-20",
"cost": {
"input": 2,
"output": 5
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2025-06"
}
},
{
"id": "magistral-small",
"name": "Magistral Small",
"provider": "mistral",
"family": "magistral-small",
"created_at": "2025-03-17 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-03-17",
"cost": {
"input": 0.5,
"output": 1.5
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2025-06"
}
},
{
"id": "magistral-small-2509",
"name": "Magistral Small 2509",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "magistral-small-latest",
"name": "Magistral Small Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "ministral-14b-2512",
"name": "Ministral 14b 2512",
"provider": "mistral",
"family": "ministral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch",
"distillation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "ministral-14b-latest",
"name": "Ministral 14b Latest",
"provider": "mistral",
"family": "ministral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch",
"distillation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "ministral-3b-2512",
"name": "Ministral 3B",
"provider": "mistral",
"family": "ministral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch",
"distillation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "ministral-3b-latest",
"name": "Ministral 3B (latest)",
"provider": "mistral",
"family": "ministral",
"created_at": "2024-10-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"distillation"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.04
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-10-04",
"cost": {
"input": 0.04,
"output": 0.04
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2024-10"
}
},
{
"id": "ministral-8b-2512",
"name": "Ministral 8B",
"provider": "mistral",
"family": "ministral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch",
"distillation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "ministral-8b-latest",
"name": "Ministral 8B (latest)",
"provider": "mistral",
"family": "ministral",
"created_at": "2024-10-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"distillation"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-10-04",
"cost": {
"input": 0.1,
"output": 0.1
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2024-10"
}
},
{
"id": "mistral-embed",
"name": "Mistral Embed",
"provider": "mistral",
"family": "mistral-embed",
"created_at": "2023-12-11 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 3072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2023-12-11",
"cost": {
"input": 0.1,
"output": 0
},
"limit": {
"context": 8000,
"output": 3072
}
}
},
{
"id": "mistral-embed-2312",
"name": "Mistral Embed",
"provider": "mistral",
"family": "mistral-embed",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-large-2411",
"name": "Mistral Large 2.1",
"provider": "mistral",
"family": "mistral-large",
"created_at": "2024-11-01 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-11-04",
"cost": {
"input": 2,
"output": 6
},
"limit": {
"context": 131072,
"output": 16384
},
"knowledge": "2024-11"
}
},
{
"id": "mistral-large-2512",
"name": "Mistral Large 3",
"provider": "mistral",
"family": "mistral-large",
"created_at": "2024-11-01 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-12-02",
"cost": {
"input": 0.5,
"output": 1.5
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2024-11"
}
},
{
"id": "mistral-large-latest",
"name": "Mistral Large (latest)",
"provider": "mistral",
"family": "mistral-large",
"created_at": "2024-11-01 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-12-02",
"cost": {
"input": 0.5,
"output": 1.5
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2024-11"
}
},
{
"id": "mistral-large-pixtral-2411",
"name": "Mistral Large",
"provider": "mistral",
"family": "mistral-large",
"created_at": "2024-11-12 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-medium",
"name": "Mistral Medium",
"provider": "mistral",
"family": "mistral-medium",
"created_at": "2025-05-05 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-medium-2505",
"name": "Mistral Medium 3",
"provider": "mistral",
"family": "mistral-medium",
"created_at": "2025-05-07 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-07",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2025-05"
}
},
{
"id": "mistral-medium-2508",
"name": "Mistral Medium 3.1",
"provider": "mistral",
"family": "mistral-medium",
"created_at": "2025-08-12 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-12",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-05"
}
},
{
"id": "mistral-medium-2604",
"name": "Mistral Medium 3.5",
"provider": "mistral",
"family": "mistral-medium",
"created_at": "2026-04-29 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.5,
"output_per_million": 7.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-29",
"cost": {
"input": 1.5,
"output": 7.5
},
"limit": {
"context": 262144,
"output": 262144
}
}
},
{
"id": "mistral-medium-3",
"name": "Mistral Medium",
"provider": "mistral",
"family": "mistral-medium",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-medium-3-5",
"name": "Mistral Medium",
"provider": "mistral",
"family": "mistral-medium",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-medium-3.5",
"name": "Mistral Medium",
"provider": "mistral",
"family": "mistral-medium",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-medium-c21211-r0-75",
"name": "Mistral Medium",
"provider": "mistral",
"family": "mistral-medium",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch",
"fine_tuning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-medium-latest",
"name": "Mistral Medium (latest)",
"provider": "mistral",
"family": "mistral-medium",
"created_at": "2026-04-29 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.5,
"output_per_million": 7.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-29",
"cost": {
"input": 1.5,
"output": 7.5
},
"limit": {
"context": 262144,
"output": 262144
}
}
},
{
"id": "mistral-moderation-2411",
"name": "Mistral Moderation",
"provider": "mistral",
"family": "mistral-moderation",
"created_at": "2024-11-26 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"moderation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-moderation-2603",
"name": "Mistral Moderation",
"provider": "mistral",
"family": "mistral-moderation",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"moderation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-moderation-latest",
"name": "Mistral Moderation",
"provider": "mistral",
"family": "mistral-moderation",
"created_at": "2024-11-26 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"moderation"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-nemo",
"name": "Mistral Nemo",
"provider": "mistral",
"family": "mistral-nemo",
"created_at": "2024-07-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-07-01",
"cost": {
"input": 0.15,
"output": 0.15
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2024-07"
}
},
{
"id": "mistral-ocr-2505",
"name": "Mistral Ocr 2505",
"provider": "mistral",
"family": "mistral",
"created_at": "2025-05-22 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-ocr-2512",
"name": "Mistral Ocr 2512",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-ocr-latest",
"name": "Mistral Ocr Latest",
"provider": "mistral",
"family": "mistral",
"created_at": "2025-05-22 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-small-2506",
"name": "Mistral Small 3.2",
"provider": "mistral",
"family": "mistral-small",
"created_at": "2025-06-20 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.3
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-06-20",
"cost": {
"input": 0.1,
"output": 0.3
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2025-03"
}
},
{
"id": "mistral-small-2603",
"name": "Mistral Small 4",
"provider": "mistral",
"family": "mistral-small",
"created_at": "2026-03-16 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-16",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 256000,
"output": 256000
},
"knowledge": "2025-06"
}
},
{
"id": "mistral-small-latest",
"name": "Mistral Small (latest)",
"provider": "mistral",
"family": "mistral-small",
"created_at": "2026-03-16 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output",
"batch",
"fine_tuning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-16",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 256000,
"output": 256000
},
"knowledge": "2025-06"
}
},
{
"id": "mistral-tiny-2407",
"name": "Mistral Tiny 2407",
"provider": "mistral",
"family": "mistral",
"created_at": "2024-07-17 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-tiny-latest",
"name": "Mistral Tiny Latest",
"provider": "mistral",
"family": "mistral",
"created_at": "2024-07-17 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-vibe-cli-fast",
"name": "Mistral Vibe Cli Fast",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-vibe-cli-latest",
"name": "Mistral Vibe Cli Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "mistral-vibe-cli-with-tools",
"name": "Mistral Vibe Cli With Tools",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "open-mistral-7b",
"name": "Mistral 7B",
"provider": "mistral",
"family": "mistral",
"created_at": "2023-09-27 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 0.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2023-09-27",
"cost": {
"input": 0.25,
"output": 0.25
},
"limit": {
"context": 8000,
"output": 8000
},
"knowledge": "2023-12"
}
},
{
"id": "open-mistral-nemo",
"name": "Open Mistral Nemo",
"provider": "mistral",
"family": "mistral",
"created_at": "2024-07-17 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "open-mistral-nemo-2407",
"name": "Open Mistral Nemo 2407",
"provider": "mistral",
"family": "mistral",
"created_at": "2024-07-17 23:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "open-mixtral-8x22b",
"name": "Mixtral 8x22B",
"provider": "mistral",
"family": "mixtral",
"created_at": "2024-04-17 00:00:00 UTC",
"context_window": 64000,
"max_output_tokens": 64000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-04-17",
"cost": {
"input": 2,
"output": 6
},
"limit": {
"context": 64000,
"output": 64000
},
"knowledge": "2024-04"
}
},
{
"id": "open-mixtral-8x7b",
"name": "Mixtral 8x7B",
"provider": "mistral",
"family": "mixtral",
"created_at": "2023-12-11 00:00:00 UTC",
"context_window": 32000,
"max_output_tokens": 32000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7,
"output_per_million": 0.7
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2023-12-11",
"cost": {
"input": 0.7,
"output": 0.7
},
"limit": {
"context": 32000,
"output": 32000
},
"knowledge": "2024-01"
}
},
{
"id": "pixtral-12b",
"name": "Pixtral 12B",
"provider": "mistral",
"family": "pixtral",
"created_at": "2024-09-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-01",
"cost": {
"input": 0.15,
"output": 0.15
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2024-09"
}
},
{
"id": "pixtral-large-2411",
"name": "Pixtral Large",
"provider": "mistral",
"family": "pixtral",
"created_at": "2024-11-12 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision",
"batch"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "pixtral-large-latest",
"name": "Pixtral Large (latest)",
"provider": "mistral",
"family": "pixtral",
"created_at": "2024-11-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming",
"structured_output",
"batch"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6
}
}
},
"metadata": {
"object": "model",
"owned_by": "mistralai",
"source": "models.dev",
"provider_id": "mistral",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-11-04",
"cost": {
"input": 2,
"output": 6
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2024-11"
}
},
{
"id": "voxtral-mini-2507",
"name": "Voxtral Mini 2507",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-2602",
"name": "Voxtral Mini 2602",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-latest",
"name": "Voxtral Mini Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-realtime-2602",
"name": "Voxtral Mini Realtime 2602",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-realtime-latest",
"name": "Voxtral Mini Realtime Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-transcribe-2507",
"name": "Voxtral Mini Transcribe 2507",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"transcription"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-transcribe-realtime-2602",
"name": "Voxtral Mini Transcribe Realtime 2602",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"transcription"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-tts-2603",
"name": "Voxtral Mini Tts 2603",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-mini-tts-latest",
"name": "Voxtral Mini Tts Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-small-2507",
"name": "Voxtral Small 2507",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "voxtral-small-latest",
"name": "Voxtral Small Latest",
"provider": "mistral",
"family": "mistral",
"created_at": null,
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "mistralai"
}
},
{
"id": "babbage-002",
"name": "babbage-002",
"provider": "openai",
"family": null,
"created_at": "2023-08-21 16:16:55 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 0.4
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "chat-latest",
"name": "chat-latest",
"provider": "openai",
"family": null,
"created_at": "2026-05-02 06:50:02 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "chatgpt-image-latest",
"name": "chatgpt-image-latest",
"provider": "openai",
"family": "gpt-image",
"created_at": "2025-12-16 00:00:00 UTC",
"context_window": 0,
"max_output_tokens": 0,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-16",
"limit": {
"context": 0,
"input": 0,
"output": 0
}
}
},
{
"id": "computer-use-preview",
"name": "computer-use-preview",
"provider": "openai",
"family": null,
"created_at": "2024-12-20 00:47:57 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "computer-use-preview-2025-03-11",
"name": "computer-use-preview-2025-03-11",
"provider": "openai",
"family": null,
"created_at": "2025-03-07 19:50:21 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "dall-e-2",
"name": "dall-e-2",
"provider": "openai",
"family": null,
"created_at": "2023-11-01 00:22:57 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "dall-e-3",
"name": "dall-e-3",
"provider": "openai",
"family": null,
"created_at": "2023-10-31 20:46:29 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "davinci-002",
"name": "davinci-002",
"provider": "openai",
"family": null,
"created_at": "2023-08-21 16:11:41 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 2.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-3.5-turbo",
"name": "GPT-3.5-turbo",
"provider": "openai",
"family": "gpt",
"created_at": "2023-03-01 00:00:00 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": "2021-09-01",
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2023-11-06",
"cost": {
"input": 0.5,
"output": 1.5,
"cache_read": 1.25
},
"limit": {
"context": 16385,
"output": 4096
},
"knowledge": "2021-09-01"
}
},
{
"id": "gpt-3.5-turbo-0125",
"name": "gpt-3.5-turbo-0125",
"provider": "openai",
"family": null,
"created_at": "2024-01-23 22:19:18 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-3.5-turbo-1106",
"name": "gpt-3.5-turbo-1106",
"provider": "openai",
"family": null,
"created_at": "2023-11-02 21:15:48 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-3.5-turbo-16k",
"name": "gpt-3.5-turbo-16k",
"provider": "openai",
"family": null,
"created_at": "2023-05-10 22:35:02 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai-internal"
}
},
{
"id": "gpt-3.5-turbo-instruct",
"name": "gpt-3.5-turbo-instruct",
"provider": "openai",
"family": null,
"created_at": "2023-08-24 18:23:47 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-3.5-turbo-instruct-0914",
"name": "gpt-3.5-turbo-instruct-0914",
"provider": "openai",
"family": null,
"created_at": "2023-09-07 21:34:32 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4",
"name": "GPT-4",
"provider": "openai",
"family": "gpt",
"created_at": "2023-11-06 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30,
"output_per_million": 60
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-04-09",
"cost": {
"input": 30,
"output": 60
},
"limit": {
"context": 8192,
"output": 8192
},
"knowledge": "2023-11"
}
},
{
"id": "gpt-4-0613",
"name": "gpt-4-0613",
"provider": "openai",
"family": null,
"created_at": "2023-06-12 16:54:56 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai"
}
},
{
"id": "gpt-4-turbo",
"name": "GPT-4 Turbo",
"provider": "openai",
"family": "gpt",
"created_at": "2023-11-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10,
"output_per_million": 30
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-04-09",
"cost": {
"input": 10,
"output": 30
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-12"
}
},
{
"id": "gpt-4-turbo-2024-04-09",
"name": "gpt-4-turbo-2024-04-09",
"provider": "openai",
"family": null,
"created_at": "2024-04-08 18:41:17 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4.1",
"name": "GPT-4.1",
"provider": "openai",
"family": "gpt",
"created_at": "2025-04-14 00:00:00 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 8,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-14",
"cost": {
"input": 2,
"output": 8,
"cache_read": 0.5
},
"limit": {
"context": 1047576,
"output": 32768
},
"knowledge": "2024-04"
}
},
{
"id": "gpt-4.1-2025-04-14",
"name": "gpt-4.1-2025-04-14",
"provider": "openai",
"family": null,
"created_at": "2025-04-10 20:09:06 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4.1-mini",
"name": "GPT-4.1 mini",
"provider": "openai",
"family": "gpt-mini",
"created_at": "2025-04-14 00:00:00 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 1.6,
"cache_read_input_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-14",
"cost": {
"input": 0.4,
"output": 1.6,
"cache_read": 0.1
},
"limit": {
"context": 1047576,
"output": 32768
},
"knowledge": "2024-04"
}
},
{
"id": "gpt-4.1-mini-2025-04-14",
"name": "gpt-4.1-mini-2025-04-14",
"provider": "openai",
"family": null,
"created_at": "2025-04-10 20:39:07 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 1.6,
"cache_read_input_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4.1-nano",
"name": "GPT-4.1 nano",
"provider": "openai",
"family": "gpt-nano",
"created_at": "2025-04-14 00:00:00 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.03
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-14",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.03
},
"limit": {
"context": 1047576,
"output": 32768
},
"knowledge": "2024-04"
}
},
{
"id": "gpt-4.1-nano-2025-04-14",
"name": "gpt-4.1-nano-2025-04-14",
"provider": "openai",
"family": null,
"created_at": "2025-04-10 21:37:05 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o",
"name": "GPT-4o",
"provider": "openai",
"family": "gpt",
"created_at": "2024-05-13 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-08-06",
"cost": {
"input": 2.5,
"output": 10,
"cache_read": 1.25
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2023-09"
}
},
{
"id": "gpt-4o-2024-05-13",
"name": "GPT-4o (2024-05-13)",
"provider": "openai",
"family": "gpt",
"created_at": "2024-05-13 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 15
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-05-13",
"cost": {
"input": 5,
"output": 15
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2023-09"
}
},
{
"id": "gpt-4o-2024-08-06",
"name": "GPT-4o (2024-08-06)",
"provider": "openai",
"family": "gpt",
"created_at": "2024-08-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-08-06",
"cost": {
"input": 2.5,
"output": 10,
"cache_read": 1.25
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2023-09"
}
},
{
"id": "gpt-4o-2024-11-20",
"name": "GPT-4o (2024-11-20)",
"provider": "openai",
"family": "gpt",
"created_at": "2024-11-20 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-11-20",
"cost": {
"input": 2.5,
"output": 10,
"cache_read": 1.25
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2023-09"
}
},
{
"id": "gpt-4o-audio-preview",
"name": "gpt-4o-audio-preview",
"provider": "openai",
"family": null,
"created_at": "2024-09-27 18:07:23 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-audio-preview-2024-12-17",
"name": "gpt-4o-audio-preview-2024-12-17",
"provider": "openai",
"family": null,
"created_at": "2024-12-12 20:10:39 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-audio-preview-2025-06-03",
"name": "gpt-4o-audio-preview-2025-06-03",
"provider": "openai",
"family": null,
"created_at": "2025-06-02 23:54:58 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini",
"name": "GPT-4o mini",
"provider": "openai",
"family": "gpt-mini",
"created_at": "2024-07-18 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.08
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-07-18",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.08
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2023-09"
}
},
{
"id": "gpt-4o-mini-2024-07-18",
"name": "gpt-4o-mini-2024-07-18",
"provider": "openai",
"family": null,
"created_at": "2024-07-16 23:31:57 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-audio-preview",
"name": "gpt-4o-mini-audio-preview",
"provider": "openai",
"family": null,
"created_at": "2024-12-16 22:17:04 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-audio-preview-2024-12-17",
"name": "gpt-4o-mini-audio-preview-2024-12-17",
"provider": "openai",
"family": null,
"created_at": "2024-12-13 18:52:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-realtime-preview",
"name": "gpt-4o-mini-realtime-preview",
"provider": "openai",
"family": null,
"created_at": "2024-12-16 22:16:20 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.4
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-realtime-preview-2024-12-17",
"name": "gpt-4o-mini-realtime-preview-2024-12-17",
"provider": "openai",
"family": null,
"created_at": "2024-12-13 17:56:41 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.4
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-search-preview",
"name": "gpt-4o-mini-search-preview",
"provider": "openai",
"family": null,
"created_at": "2025-03-07 23:46:01 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-search-preview-2025-03-11",
"name": "gpt-4o-mini-search-preview-2025-03-11",
"provider": "openai",
"family": null,
"created_at": "2025-03-07 23:40:58 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-transcribe",
"name": "gpt-4o-mini-transcribe",
"provider": "openai",
"family": null,
"created_at": "2025-03-15 19:56:36 UTC",
"context_window": 16000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-transcribe-2025-03-20",
"name": "gpt-4o-mini-transcribe-2025-03-20",
"provider": "openai",
"family": null,
"created_at": "2025-12-13 07:22:25 UTC",
"context_window": 16000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-transcribe-2025-12-15",
"name": "gpt-4o-mini-transcribe-2025-12-15",
"provider": "openai",
"family": null,
"created_at": "2025-12-13 07:20:07 UTC",
"context_window": 16000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-tts",
"name": "gpt-4o-mini-tts",
"provider": "openai",
"family": null,
"created_at": "2025-03-19 17:05:59 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 12.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-tts-2025-03-20",
"name": "gpt-4o-mini-tts-2025-03-20",
"provider": "openai",
"family": null,
"created_at": "2025-12-13 07:25:31 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 12.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-mini-tts-2025-12-15",
"name": "gpt-4o-mini-tts-2025-12-15",
"provider": "openai",
"family": null,
"created_at": "2025-12-13 07:27:17 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 12.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-realtime-preview",
"name": "gpt-4o-realtime-preview",
"provider": "openai",
"family": null,
"created_at": "2024-09-30 01:33:18 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 20.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-realtime-preview-2024-12-17",
"name": "gpt-4o-realtime-preview-2024-12-17",
"provider": "openai",
"family": null,
"created_at": "2024-12-11 19:30:30 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 20.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-realtime-preview-2025-06-03",
"name": "gpt-4o-realtime-preview-2025-06-03",
"provider": "openai",
"family": null,
"created_at": "2025-06-02 23:43:58 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 20.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-search-preview",
"name": "gpt-4o-search-preview",
"provider": "openai",
"family": null,
"created_at": "2026-02-24 03:58:54 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-search-preview-2025-03-11",
"name": "gpt-4o-search-preview-2025-03-11",
"provider": "openai",
"family": null,
"created_at": "2026-02-24 04:00:21 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-transcribe",
"name": "gpt-4o-transcribe",
"provider": "openai",
"family": null,
"created_at": "2025-03-15 19:54:23 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-4o-transcribe-diarize",
"name": "gpt-4o-transcribe-diarize",
"provider": "openai",
"family": null,
"created_at": "2025-06-24 21:01:27 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5",
"name": "GPT-5",
"provider": "openai",
"family": "gpt",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-08-07",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5-2025-08-07",
"name": "gpt-5-2025-08-07",
"provider": "openai",
"family": null,
"created_at": "2025-08-01 19:09:20 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5-chat-latest",
"name": "GPT-5 Chat (latest)",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"reasoning",
"vision",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-07",
"cost": {
"input": 1.25,
"output": 10
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5-codex",
"name": "GPT-5-Codex",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-09-15 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-09-15",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5-mini",
"name": "GPT-5 Mini",
"provider": "openai",
"family": "gpt-mini",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-05-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-08-07",
"cost": {
"input": 0.25,
"output": 2,
"cache_read": 0.025
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-05-30"
}
},
{
"id": "gpt-5-mini-2025-08-07",
"name": "gpt-5-mini-2025-08-07",
"provider": "openai",
"family": null,
"created_at": "2025-08-05 20:31:07 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5-nano",
"name": "GPT-5 Nano",
"provider": "openai",
"family": "gpt-nano",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-05-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.005
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-08-07",
"cost": {
"input": 0.05,
"output": 0.4,
"cache_read": 0.005
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-05-30"
}
},
{
"id": "gpt-5-nano-2025-08-07",
"name": "gpt-5-nano-2025-08-07",
"provider": "openai",
"family": null,
"created_at": "2025-08-05 20:38:23 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.005
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5-pro",
"name": "GPT-5 Pro",
"provider": "openai",
"family": "gpt-pro",
"created_at": "2025-10-06 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 272000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 120
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-10-06",
"cost": {
"input": 15,
"output": 120
},
"limit": {
"context": 400000,
"input": 272000,
"output": 272000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5-pro-2025-10-06",
"name": "gpt-5-pro-2025-10-06",
"provider": "openai",
"family": null,
"created_at": "2025-10-03 05:35:07 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5-search-api",
"name": "gpt-5-search-api",
"provider": "openai",
"family": null,
"created_at": "2025-10-03 18:03:49 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5-search-api-2025-10-14",
"name": "gpt-5-search-api-2025-10-14",
"provider": "openai",
"family": null,
"created_at": "2025-10-09 21:06:00 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.1",
"name": "GPT-5.1",
"provider": "openai",
"family": "gpt",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.13
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.13
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5.1-2025-11-13",
"name": "gpt-5.1-2025-11-13",
"provider": "openai",
"family": null,
"created_at": "2025-11-10 18:45:53 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.1-chat-latest",
"name": "GPT-5.1 Chat",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5.1-codex",
"name": "GPT-5.1 Codex",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5.1-codex-max",
"name": "GPT-5.1 Codex Max",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5.1-codex-mini",
"name": "GPT-5.1 Codex mini",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-11-13",
"cost": {
"input": 0.25,
"output": 2,
"cache_read": 0.025
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "gpt-5.2",
"name": "GPT-5.2",
"provider": "openai",
"family": "gpt",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.2-2025-12-11",
"name": "gpt-5.2-2025-12-11",
"provider": "openai",
"family": null,
"created_at": "2025-12-09 20:43:48 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.2-chat-latest",
"name": "GPT-5.2 Chat",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.2-codex",
"name": "GPT-5.2 Codex",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.2-pro",
"name": "GPT-5.2 Pro",
"provider": "openai",
"family": "gpt-pro",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 21,
"output_per_million": 168
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 21,
"output": 168
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.2-pro-2025-12-11",
"name": "gpt-5.2-pro-2025-12-11",
"provider": "openai",
"family": null,
"created_at": "2025-12-10 05:19:19 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.3-chat-latest",
"name": "GPT-5.3 Chat (latest)",
"provider": "openai",
"family": "gpt",
"created_at": "2026-03-03 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-03",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.3-codex",
"name": "GPT-5.3 Codex",
"provider": "openai",
"family": "gpt-codex",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-02-05",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.3-codex-spark",
"name": "GPT-5.3 Codex Spark",
"provider": "openai",
"family": "gpt-codex-spark",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-02-05",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 128000,
"input": 100000,
"output": 32000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.4",
"name": "GPT-5.4",
"provider": "openai",
"family": "gpt",
"created_at": "2026-03-05 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 15,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-03-05",
"cost": {
"input": 2.5,
"output": 15,
"cache_read": 0.25,
"context_over_200k": {
"input": 5,
"output": 22.5,
"cache_read": 0.5
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.4-2026-03-05",
"name": "gpt-5.4-2026-03-05",
"provider": "openai",
"family": null,
"created_at": "2026-03-04 19:54:22 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.4-mini",
"name": "GPT-5.4 mini",
"provider": "openai",
"family": "gpt-mini",
"created_at": "2026-03-17 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.75,
"output_per_million": 4.5,
"cache_read_input_per_million": 0.075
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-03-17",
"cost": {
"input": 0.75,
"output": 4.5,
"cache_read": 0.075
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.4-mini-2026-03-17",
"name": "gpt-5.4-mini-2026-03-17",
"provider": "openai",
"family": null,
"created_at": "2026-03-14 01:17:56 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.4-nano",
"name": "GPT-5.4 nano",
"provider": "openai",
"family": "gpt-nano",
"created_at": "2026-03-17 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 1.25,
"cache_read_input_per_million": 0.02
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-03-17",
"cost": {
"input": 0.2,
"output": 1.25,
"cache_read": 0.02
},
"limit": {
"context": 400000,
"input": 272000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.4-nano-2026-03-17",
"name": "gpt-5.4-nano-2026-03-17",
"provider": "openai",
"family": null,
"created_at": "2026-03-14 01:13:57 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.005
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.4-pro",
"name": "GPT-5.4 Pro",
"provider": "openai",
"family": "gpt-pro",
"created_at": "2026-03-05 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30,
"output_per_million": 180
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-03-05",
"cost": {
"input": 30,
"output": 180,
"context_over_200k": {
"input": 60,
"output": 270
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "gpt-5.4-pro-2026-03-05",
"name": "gpt-5.4-pro-2026-03-05",
"provider": "openai",
"family": null,
"created_at": "2026-03-04 21:27:37 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.5",
"name": "GPT-5.5",
"provider": "openai",
"family": "gpt",
"created_at": "2026-04-23 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-12-01",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 30,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-23",
"cost": {
"input": 5,
"output": 30,
"cache_read": 0.5,
"context_over_200k": {
"input": 10,
"output": 45,
"cache_read": 1
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-12-01"
}
},
{
"id": "gpt-5.5-2026-04-23",
"name": "gpt-5.5-2026-04-23",
"provider": "openai",
"family": null,
"created_at": "2026-04-22 06:27:21 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-5.5-pro",
"name": "GPT-5.5 Pro",
"provider": "openai",
"family": "gpt-pro",
"created_at": "2026-04-23 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-12-01",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30,
"output_per_million": 180
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-23",
"cost": {
"input": 30,
"output": 180,
"context_over_200k": {
"input": 60,
"output": 270
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-12-01"
}
},
{
"id": "gpt-5.5-pro-2026-04-23",
"name": "gpt-5.5-pro-2026-04-23",
"provider": "openai",
"family": null,
"created_at": "2026-04-22 21:47:50 UTC",
"context_window": 128000,
"max_output_tokens": 400000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-audio",
"name": "gpt-audio",
"provider": "openai",
"family": null,
"created_at": "2025-08-28 00:00:49 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-audio-1.5",
"name": "gpt-audio-1.5",
"provider": "openai",
"family": null,
"created_at": "2026-02-20 01:28:05 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-audio-2025-08-28",
"name": "gpt-audio-2025-08-28",
"provider": "openai",
"family": null,
"created_at": "2025-08-27 00:55:46 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-audio-mini",
"name": "gpt-audio-mini",
"provider": "openai",
"family": null,
"created_at": "2025-10-03 17:20:27 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-audio-mini-2025-10-06",
"name": "gpt-audio-mini-2025-10-06",
"provider": "openai",
"family": null,
"created_at": "2025-10-03 17:22:17 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-audio-mini-2025-12-15",
"name": "gpt-audio-mini-2025-12-15",
"provider": "openai",
"family": null,
"created_at": "2025-12-15 00:53:28 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-image-1",
"name": "gpt-image-1",
"provider": "openai",
"family": "gpt-image",
"created_at": "2025-04-24 00:00:00 UTC",
"context_window": 0,
"max_output_tokens": 0,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"cache_read_input_per_million": 1.25
}
},
"images": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 40.0,
"cache_read_input_per_million": 2.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-04-24",
"limit": {
"context": 0,
"input": 0,
"output": 0
}
}
},
{
"id": "gpt-image-1-mini",
"name": "gpt-image-1-mini",
"provider": "openai",
"family": "gpt-image",
"created_at": "2025-09-26 00:00:00 UTC",
"context_window": 0,
"max_output_tokens": 0,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"cache_read_input_per_million": 0.2
}
},
"images": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-09-26",
"limit": {
"context": 0,
"input": 0,
"output": 0
}
}
},
{
"id": "gpt-image-1.5",
"name": "gpt-image-1.5",
"provider": "openai",
"family": "gpt-image",
"created_at": "2025-11-25 00:00:00 UTC",
"context_window": 0,
"max_output_tokens": 0,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"cache_read_input_per_million": 1.25
}
},
"images": {
"standard": {
"input_per_million": 8.0,
"output_per_million": 32.0,
"cache_read_input_per_million": 2.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-11-25",
"limit": {
"context": 0,
"input": 0,
"output": 0
}
}
},
{
"id": "gpt-image-2",
"name": "gpt-image-2",
"provider": "openai",
"family": null,
"created_at": "2026-04-17 04:23:15 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-image-2-2026-04-21",
"name": "gpt-image-2-2026-04-21",
"provider": "openai",
"family": null,
"created_at": "2026-04-17 04:26:34 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-realtime",
"name": "gpt-realtime",
"provider": "openai",
"family": null,
"created_at": "2025-08-27 05:15:01 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-realtime-1.5",
"name": "gpt-realtime-1.5",
"provider": "openai",
"family": null,
"created_at": "2026-02-19 00:37:49 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-realtime-2025-08-28",
"name": "gpt-realtime-2025-08-28",
"provider": "openai",
"family": null,
"created_at": "2025-08-27 05:16:13 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-realtime-mini",
"name": "gpt-realtime-mini",
"provider": "openai",
"family": null,
"created_at": "2025-10-03 18:45:33 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-realtime-mini-2025-10-06",
"name": "gpt-realtime-mini-2025-10-06",
"provider": "openai",
"family": null,
"created_at": "2025-10-03 18:46:15 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "gpt-realtime-mini-2025-12-15",
"name": "gpt-realtime-mini-2025-12-15",
"provider": "openai",
"family": null,
"created_at": "2025-12-13 07:46:47 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o1",
"name": "o1",
"provider": "openai",
"family": "o",
"created_at": "2024-12-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 60,
"cache_read_input_per_million": 7.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2024-12-05",
"cost": {
"input": 15,
"output": 60,
"cache_read": 7.5
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2023-09"
}
},
{
"id": "o1-2024-12-17",
"name": "o1-2024-12-17",
"provider": "openai",
"family": null,
"created_at": "2024-12-16 05:29:36 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15.0,
"output_per_million": 60.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o1-mini",
"name": "o1-mini",
"provider": "openai",
"family": "o-mini",
"created_at": "2024-09-12 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.55
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2024-09-12",
"cost": {
"input": 1.1,
"output": 4.4,
"cache_read": 0.55
},
"limit": {
"context": 128000,
"output": 65536
},
"knowledge": "2023-09"
}
},
{
"id": "o1-preview",
"name": "o1-preview",
"provider": "openai",
"family": "o",
"created_at": "2024-09-12 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 60,
"cache_read_input_per_million": 7.5
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2024-09-12",
"cost": {
"input": 15,
"output": 60,
"cache_read": 7.5
},
"limit": {
"context": 128000,
"output": 32768
},
"knowledge": "2023-09"
}
},
{
"id": "o1-pro",
"name": "o1-pro",
"provider": "openai",
"family": "o-pro",
"created_at": "2025-03-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 150,
"output_per_million": 600
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-03-19",
"cost": {
"input": 150,
"output": 600
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2023-09"
}
},
{
"id": "o1-pro-2025-03-19",
"name": "o1-pro-2025-03-19",
"provider": "openai",
"family": null,
"created_at": "2025-03-17 22:45:04 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 150.0,
"output_per_million": 600.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o3",
"name": "o3",
"provider": "openai",
"family": "o",
"created_at": "2025-04-16 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 8,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-04-16",
"cost": {
"input": 2,
"output": 8,
"cache_read": 0.5
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-05"
}
},
{
"id": "o3-2025-04-16",
"name": "o3-2025-04-16",
"provider": "openai",
"family": null,
"created_at": "2025-04-08 17:28:21 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o3-deep-research",
"name": "o3-deep-research",
"provider": "openai",
"family": "o",
"created_at": "2024-06-26 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10,
"output_per_million": 40,
"cache_read_input_per_million": 2.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2024-06-26",
"cost": {
"input": 10,
"output": 40,
"cache_read": 2.5
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-05"
}
},
{
"id": "o3-deep-research-2025-06-26",
"name": "o3-deep-research-2025-06-26",
"provider": "openai",
"family": null,
"created_at": "2025-06-25 15:26:59 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o3-mini",
"name": "o3-mini",
"provider": "openai",
"family": "o-mini",
"created_at": "2024-12-20 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.55
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-01-29",
"cost": {
"input": 1.1,
"output": 4.4,
"cache_read": 0.55
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-05"
}
},
{
"id": "o3-mini-2025-01-31",
"name": "o3-mini-2025-01-31",
"provider": "openai",
"family": null,
"created_at": "2025-01-27 20:36:40 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o3-pro",
"name": "o3-pro",
"provider": "openai",
"family": "o-pro",
"created_at": "2025-06-10 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 20,
"output_per_million": 80
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-06-10",
"cost": {
"input": 20,
"output": 80
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-05"
}
},
{
"id": "o3-pro-2025-06-10",
"name": "o3-pro-2025-06-10",
"provider": "openai",
"family": null,
"created_at": "2025-06-05 23:39:21 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o4-mini",
"name": "o4-mini",
"provider": "openai",
"family": "o-mini",
"created_at": "2025-04-16 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.28
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-04-16",
"cost": {
"input": 1.1,
"output": 4.4,
"cache_read": 0.28
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-05"
}
},
{
"id": "o4-mini-2025-04-16",
"name": "o4-mini-2025-04-16",
"provider": "openai",
"family": null,
"created_at": "2025-04-08 17:31:46 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "o4-mini-deep-research",
"name": "o4-mini-deep-research",
"provider": "openai",
"family": "o-mini",
"created_at": "2024-06-26 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 8,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2024-06-26",
"cost": {
"input": 2,
"output": 8,
"cache_read": 0.5
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-05"
}
},
{
"id": "o4-mini-deep-research-2025-06-26",
"name": "o4-mini-deep-research-2025-06-26",
"provider": "openai",
"family": null,
"created_at": "2025-06-25 15:42:01 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "omni-moderation-2024-09-26",
"name": "omni-moderation-2024-09-26",
"provider": "openai",
"family": null,
"created_at": "2024-11-27 19:07:46 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "omni-moderation-latest",
"name": "omni-moderation-latest",
"provider": "openai",
"family": null,
"created_at": "2024-11-15 16:47:45 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "sora-2",
"name": "sora-2",
"provider": "openai",
"family": null,
"created_at": "2025-10-05 23:56:55 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "sora-2-pro",
"name": "sora-2-pro",
"provider": "openai",
"family": null,
"created_at": "2025-10-05 23:57:43 UTC",
"context_window": 4096,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "text-embedding-3-large",
"name": "text-embedding-3-large",
"provider": "openai",
"family": "text-embedding",
"created_at": "2024-01-25 00:00:00 UTC",
"context_window": 8191,
"max_output_tokens": 3072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.13
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2024-01-25",
"cost": {
"input": 0.13,
"output": 0
},
"limit": {
"context": 8191,
"output": 3072
},
"knowledge": "2024-01"
}
},
{
"id": "text-embedding-3-small",
"name": "text-embedding-3-small",
"provider": "openai",
"family": "text-embedding",
"created_at": "2024-01-25 00:00:00 UTC",
"context_window": 8191,
"max_output_tokens": 1536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.02
}
}
},
"metadata": {
"object": "model",
"owned_by": "system",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2024-01-25",
"cost": {
"input": 0.02,
"output": 0
},
"limit": {
"context": 8191,
"output": 1536
},
"knowledge": "2024-01"
}
},
{
"id": "text-embedding-ada-002",
"name": "text-embedding-ada-002",
"provider": "openai",
"family": "text-embedding",
"created_at": "2022-12-15 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 1536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai-internal",
"source": "models.dev",
"provider_id": "openai",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2022-12-15",
"cost": {
"input": 0.1,
"output": 0
},
"limit": {
"context": 8192,
"output": 1536
},
"knowledge": "2022-12"
}
},
{
"id": "tts-1",
"name": "tts-1",
"provider": "openai",
"family": null,
"created_at": "2023-04-19 21:49:11 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15.0,
"output_per_million": 15.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai-internal"
}
},
{
"id": "tts-1-1106",
"name": "tts-1-1106",
"provider": "openai",
"family": null,
"created_at": "2023-11-03 23:14:01 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15.0,
"output_per_million": 15.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "tts-1-hd",
"name": "tts-1-hd",
"provider": "openai",
"family": null,
"created_at": "2023-11-03 21:13:35 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "tts-1-hd-1106",
"name": "tts-1-hd-1106",
"provider": "openai",
"family": null,
"created_at": "2023-11-03 23:18:53 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"object": "model",
"owned_by": "system"
}
},
{
"id": "whisper-1",
"name": "whisper-1",
"provider": "openai",
"family": null,
"created_at": "2023-02-27 21:13:04 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.006,
"output_per_million": 0.006
}
}
},
"metadata": {
"object": "model",
"owned_by": "openai-internal"
}
},
{
"id": "ai21/jamba-large-1.7",
"name": "AI21: Jamba Large 1.7",
"provider": "openrouter",
"family": "ai21",
"created_at": "2025-08-08 16:03:40 UTC",
"context_window": 256000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0
}
}
},
"metadata": {
"description": "Jamba Large 1.7 is the latest model in the Jamba open family, offering improvements in grounding, instruction-following, and overall efficiency. Built on a hybrid SSM-Transformer architecture with a 256K context...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 4096,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "aion-labs/aion-1.0",
"name": "AionLabs: Aion-1.0",
"provider": "openrouter",
"family": "aion-labs",
"created_at": "2025-02-04 19:32:37 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 4.0,
"output_per_million": 8.0
}
}
},
"metadata": {
"description": "Aion-1.0 is a multi-model system designed for high performance across various tasks, including reasoning and coding. It is built on DeepSeek-R1, augmented with additional models and techniques such as Tree...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"temperature",
"top_p"
]
}
},
{
"id": "aion-labs/aion-1.0-mini",
"name": "AionLabs: Aion-1.0-Mini",
"provider": "openrouter",
"family": "aion-labs",
"created_at": "2025-02-04 19:25:07 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7,
"output_per_million": 1.4
}
}
},
"metadata": {
"description": "Aion-1.0-Mini 32B parameter model is a distilled version of the DeepSeek-R1 model, designed for strong performance in reasoning domains such as mathematics, coding, and logic. It is a modified variant...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"temperature",
"top_p"
]
}
},
{
"id": "aion-labs/aion-2.0",
"name": "AionLabs: Aion-2.0",
"provider": "openrouter",
"family": "aion-labs",
"created_at": "2026-02-23 21:15:06 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7999999999999999,
"output_per_million": 1.5999999999999999,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Aion-2.0 is a variant of DeepSeek V3.2 optimized for immersive roleplaying and storytelling. It is particularly strong at introducing tension, crises, and conflict into stories, making narratives feel more engaging....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"temperature",
"top_p"
]
}
},
{
"id": "aion-labs/aion-rp-llama-3.1-8b",
"name": "AionLabs: Aion-RP 1.0 (8B)",
"provider": "openrouter",
"family": "aion-labs",
"created_at": "2025-02-04 19:18:38 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7999999999999999,
"output_per_million": 1.5999999999999999
}
}
},
"metadata": {
"description": "Aion-RP-Llama-3.1-8B ranks the highest in the character evaluation portion of the RPBench-Auto benchmark, a roleplaying-specific variant of Arena-Hard-Auto, where LLMs evaluate each other’s responses. It is a fine-tuned base model...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"temperature",
"top_p"
]
}
},
{
"id": "alfredpros/codellama-7b-instruct-solidity",
"name": "AlfredPros: CodeLLaMa 7B Instruct Solidity",
"provider": "openrouter",
"family": "alfredpros",
"created_at": "2025-04-14 14:44:34 UTC",
"context_window": 4096,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7999999999999999,
"output_per_million": 1.2
}
}
},
"metadata": {
"description": "A finetuned 7 billion parameters Code LLaMA - Instruct model to generate Solidity smart contract using 4-bit QLoRA finetuning provided by PEFT library.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": "alpaca"
},
"top_provider": {
"context_length": 4096,
"max_completion_tokens": 4096,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "alibaba/tongyi-deepresearch-30b-a3b",
"name": "Tongyi DeepResearch 30B A3B",
"provider": "openrouter",
"family": "alibaba",
"created_at": "2025-09-18 15:53:24 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09,
"output_per_million": 0.44999999999999996,
"cache_read_input_per_million": 0.09
}
}
},
"metadata": {
"description": "Tongyi DeepResearch is an agentic large language model developed by Tongyi Lab, with 30 billion total parameters activating only 3 billion per token. It's optimized for long-horizon, deep information-seeking tasks...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "allenai/olmo-3-32b-think",
"name": "AllenAI: Olmo 3 32B Think",
"provider": "openrouter",
"family": "allenai",
"created_at": "2025-11-21 20:51:16 UTC",
"context_window": 65536,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.5
}
}
},
"metadata": {
"description": "Olmo 3 32B Think is a large-scale, 32-billion-parameter model purpose-built for deep reasoning, complex logic chains and advanced instruction-following scenarios. Its capacity enables strong performance on demanding evaluation tasks and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "allenai/olmo-3.1-32b-instruct",
"name": "AllenAI: Olmo 3.1 32B Instruct",
"provider": "openrouter",
"family": "allenai",
"created_at": "2026-01-06 19:42:34 UTC",
"context_window": 65536,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.19999999999999998,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "Olmo 3.1 32B Instruct is a large-scale, 32-billion-parameter instruction-tuned language model engineered for high-performance conversational AI, multi-turn dialogue, and practical instruction following. As part of the Olmo 3.1 family, this...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "alpindale/goliath-120b",
"name": "Goliath 120B",
"provider": "openrouter",
"family": "alpindale",
"created_at": "2023-11-10 00:00:00 UTC",
"context_window": 6144,
"max_output_tokens": 1024,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.75,
"output_per_million": 7.5
}
}
},
"metadata": {
"description": "A large LLM created by combining two fine-tuned Llama 70B models into one 120B model. Combines Xwin and Euryale. Credits to - [@chargoddard](https://huggingface.co/chargoddard) for developing the framework used to merge...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama2",
"instruct_type": "airoboros"
},
"top_provider": {
"context_length": 6144,
"max_completion_tokens": 1024,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_a",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "amazon/nova-2-lite-v1",
"name": "Amazon: Nova 2 Lite",
"provider": "openrouter",
"family": "amazon",
"created_at": "2025-12-02 17:31:12 UTC",
"context_window": 1000000,
"max_output_tokens": 65535,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5
}
}
},
"metadata": {
"description": "Nova 2 Lite is a fast, cost-effective reasoning model for everyday workloads that can process text, images, and videos to generate text. Nova 2 Lite demonstrates standout capabilities in processing...",
"architecture": {
"modality": "text+image+file+video->text",
"input_modalities": [
"text",
"image",
"video",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Nova",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65535,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "amazon/nova-lite-v1",
"name": "Amazon: Nova Lite 1.0",
"provider": "openrouter",
"family": "amazon",
"created_at": "2024-12-05 22:22:43 UTC",
"context_window": 300000,
"max_output_tokens": 5120,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.24
}
}
},
"metadata": {
"description": "Amazon Nova Lite 1.0 is a very low-cost multimodal model from Amazon that focused on fast processing of image, video, and text inputs to generate text output. Amazon Nova Lite...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Nova",
"instruct_type": null
},
"top_provider": {
"context_length": 300000,
"max_completion_tokens": 5120,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "amazon/nova-micro-v1",
"name": "Amazon: Nova Micro 1.0",
"provider": "openrouter",
"family": "amazon",
"created_at": "2024-12-05 22:20:37 UTC",
"context_window": 128000,
"max_output_tokens": 5120,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.035,
"output_per_million": 0.14
}
}
},
"metadata": {
"description": "Amazon Nova Micro 1.0 is a text-only model that delivers the lowest latency responses in the Amazon Nova family of models at a very low cost. With a context length...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Nova",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 5120,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "amazon/nova-premier-v1",
"name": "Amazon: Nova Premier 1.0",
"provider": "openrouter",
"family": "amazon",
"created_at": "2025-10-31 22:38:52 UTC",
"context_window": 1000000,
"max_output_tokens": 32000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 12.5,
"cache_read_input_per_million": 0.625
}
}
},
"metadata": {
"description": "Amazon Nova Premier is the most capable of Amazon’s multimodal models for complex reasoning tasks and for use as the best teacher for distilling custom models.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Nova",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 32000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "amazon/nova-pro-v1",
"name": "Amazon: Nova Pro 1.0",
"provider": "openrouter",
"family": "amazon",
"created_at": "2024-12-05 22:05:03 UTC",
"context_window": 300000,
"max_output_tokens": 5120,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7999999999999999,
"output_per_million": 3.1999999999999997
}
}
},
"metadata": {
"description": "Amazon Nova Pro 1.0 is a capable multimodal model from Amazon focused on providing a combination of accuracy, speed, and cost for a wide range of tasks. As of December...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Nova",
"instruct_type": null
},
"top_provider": {
"context_length": 300000,
"max_completion_tokens": 5120,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "anthracite-org/magnum-v4-72b",
"name": "Magnum v4 72B",
"provider": "openrouter",
"family": "anthracite-org",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 16384,
"max_output_tokens": 2048,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 5.0
}
}
},
"metadata": {
"description": "This is a series of models designed to replicate the prose quality of the Claude 3 models, specifically Sonnet(https://openrouter.ai/anthropic/claude-3.5-sonnet) and Opus(https://openrouter.ai/anthropic/claude-3-opus).\n\nThe model is fine-tuned on top of [Qwen2.5 72B](https://openrouter.ai/qwen/qwen-2.5-72b-instruct).",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 16384,
"max_completion_tokens": 2048,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_a",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "anthropic/claude-3-haiku",
"name": "Anthropic: Claude 3 Haiku",
"provider": "openrouter",
"family": "anthropic",
"created_at": "2024-03-13 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.25,
"cache_read_input_per_million": 0.03
}
}
},
"metadata": {
"description": "Claude 3 Haiku is Anthropic's fastest and most compact model for\nnear-instant responsiveness. Quick and accurate targeted performance.\n\nSee the launch announcement and benchmark results [here](https://www.anthropic.com/news/claude-3-haiku)\n\n#multimodal",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "anthropic/claude-3.5-haiku",
"name": "Claude Haiku 3.5",
"provider": "openrouter",
"family": "claude-haiku",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 4,
"cache_read_input_per_million": 0.08,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"description": "Claude 3.5 Haiku features offers enhanced capabilities in speed, coding accuracy, and tool use. Engineered to excel in real-time applications, it delivers quick response times that are essential for dynamic...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 8192,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 0.8,
"output": 4,
"cache_read": 0.08,
"cache_write": 1
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-07-31"
}
},
{
"id": "anthropic/claude-3.7-sonnet",
"name": "Claude Sonnet 3.7",
"provider": "openrouter",
"family": "claude-sonnet",
"created_at": "2025-02-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"description": "Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 64000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-02-19",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 128000
},
"knowledge": "2024-01"
}
},
{
"id": "anthropic/claude-3.7-sonnet:thinking",
"name": "Anthropic: Claude 3.7 Sonnet (thinking)",
"provider": "openrouter",
"family": "anthropic",
"created_at": "2025-02-24 18:35:10 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 15.0,
"cache_read_input_per_million": 0.3
}
}
},
"metadata": {
"description": "Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 64000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "anthropic/claude-haiku-4.5",
"name": "Claude Haiku 4.5",
"provider": "openrouter",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"description": "Claude Haiku 4.5 is Anthropic’s fastest and most efficient model, delivering near-frontier intelligence at a fraction of the cost and latency of larger Claude models. Matching Claude Sonnet 4’s performance...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 64000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "anthropic/claude-opus-4",
"name": "Claude Opus 4",
"provider": "openrouter",
"family": "claude-opus",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"description": "Claude Opus 4 is benchmarked as the world’s best coding model, at time of release, bringing sustained performance on complex, long-running tasks and agent workflows. It sets new benchmarks in...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 32000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic/claude-opus-4.1",
"name": "Claude Opus 4.1",
"provider": "openrouter",
"family": "claude-opus",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"description": "Claude Opus 4.1 is an updated version of Anthropic’s flagship model, offering improved performance in coding, reasoning, and agentic tasks. It achieves 74.5% on SWE-bench Verified and shows notable gains...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 32000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic/claude-opus-4.5",
"name": "Claude Opus 4.5",
"provider": "openrouter",
"family": "claude-opus",
"created_at": "2025-11-24 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-05-30",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"description": "Claude Opus 4.5 is Anthropic’s frontier reasoning model optimized for complex software engineering, agentic workflows, and long-horizon computer use. It offers strong multimodal capabilities, competitive performance across real-world coding and...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 64000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"verbosity"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-24",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-05-30"
}
},
{
"id": "anthropic/claude-opus-4.6",
"name": "Claude Opus 4.6",
"provider": "openrouter",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"description": "Opus 4.6 is Anthropic’s strongest model for coding and long-running professional tasks. It is built for agents that operate across entire workflows rather than single prompts, making it especially effective...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p",
"verbosity"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-05",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25,
"context_over_200k": {
"input": 10,
"output": 37.5,
"cache_read": 1,
"cache_write": 12.5
}
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "anthropic/claude-opus-4.6-fast",
"name": "Anthropic: Claude Opus 4.6 (Fast)",
"provider": "openrouter",
"family": "anthropic",
"created_at": "2026-04-07 20:07:52 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30.0,
"output_per_million": 150.0,
"cache_read_input_per_million": 3.0
}
}
},
"metadata": {
"description": "Fast-mode variant of [Opus 4.6](/anthropic/claude-opus-4.6) - identical capabilities with higher output speed at premium 6x pricing.\n\nLearn more in Anthropic's docs: https://platform.claude.com/docs/en/build-with-claude/fast-mode",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p",
"verbosity"
]
}
},
{
"id": "anthropic/claude-opus-4.7",
"name": "Claude Opus 4.7",
"provider": "openrouter",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"description": "Opus 4.7 is the next generation of Anthropic's Opus family, built for long-running, asynchronous agents. Building on the coding and agentic strengths of Opus 4.6, it delivers stronger performance on...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"tool_choice",
"tools",
"verbosity"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25,
"context_over_200k": {
"input": 10,
"output": 37.5,
"cache_read": 1,
"cache_write": 12.5
}
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "anthropic/claude-sonnet-4",
"name": "Claude Sonnet 4",
"provider": "openrouter",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"description": "Claude Sonnet 4 significantly enhances the capabilities of its predecessor, Sonnet 3.7, excelling in both coding and reasoning tasks with improved precision and controllability. Achieving state-of-the-art performance on SWE-bench (72.7%),...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 64000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75,
"context_over_200k": {
"input": 6,
"output": 22.5,
"cache_read": 0.6,
"cache_write": 7.5
}
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "anthropic/claude-sonnet-4.5",
"name": "Claude Sonnet 4.5",
"provider": "openrouter",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"description": "Claude Sonnet 4.5 is Anthropic’s most advanced Sonnet model to date, optimized for real-world agents and coding workflows. It delivers state-of-the-art performance on coding benchmarks such as SWE-bench Verified, with...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 64000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75,
"context_over_200k": {
"input": 6,
"output": 22.5,
"cache_read": 0.6,
"cache_write": 7.5
}
},
"limit": {
"context": 1000000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "anthropic/claude-sonnet-4.6",
"name": "Claude Sonnet 4.6",
"provider": "openrouter",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"description": "Sonnet 4.6 is Anthropic's most capable Sonnet-class model yet, with frontier performance across coding, agents, and professional work. It excels at iterative development, complex codebase navigation, end-to-end project management with...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Claude",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p",
"verbosity"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-17",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75,
"context_over_200k": {
"input": 6,
"output": 22.5,
"cache_read": 0.6,
"cache_write": 7.5
}
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "arcee-ai/coder-large",
"name": "Arcee AI: Coder Large",
"provider": "openrouter",
"family": "arcee-ai",
"created_at": "2025-05-05 20:57:43 UTC",
"context_window": 32768,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 0.7999999999999999
}
}
},
"metadata": {
"description": "Coder‑Large is a 32 B‑parameter offspring of Qwen 2.5‑Instruct that has been further trained on permissively‑licensed GitHub, CodeSearchNet and synthetic bug‑fix corpora. It supports a 32k context window, enabling multi‑file...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "arcee-ai/maestro-reasoning",
"name": "Arcee AI: Maestro Reasoning",
"provider": "openrouter",
"family": "arcee-ai",
"created_at": "2025-05-05 21:41:09 UTC",
"context_window": 131072,
"max_output_tokens": 32000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8999999999999999,
"output_per_million": 3.3000000000000003
}
}
},
"metadata": {
"description": "Maestro Reasoning is Arcee's flagship analysis model: a 32 B‑parameter derivative of Qwen 2.5‑32 B tuned with DPO and chain‑of‑thought RL for step‑by‑step logic. Compared to the earlier 7 B...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "arcee-ai/spotlight",
"name": "Arcee AI: Spotlight",
"provider": "openrouter",
"family": "arcee-ai",
"created_at": "2025-05-05 21:45:52 UTC",
"context_window": 131072,
"max_output_tokens": 65537,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.18,
"output_per_million": 0.18
}
}
},
"metadata": {
"description": "Spotlight is a 7‑billion‑parameter vision‑language model derived from Qwen 2.5‑VL and fine‑tuned by Arcee AI for tight image‑text grounding tasks. It offers a 32 k‑token context window, enabling rich multimodal...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 65537,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "arcee-ai/trinity-large-preview",
"name": "Arcee AI: Trinity Large Preview",
"provider": "openrouter",
"family": "arcee-ai",
"created_at": "2026-01-27 22:24:30 UTC",
"context_window": 131000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.44999999999999996
}
}
},
"metadata": {
"description": "Trinity-Large-Preview is a frontier-scale open-weight language model from Arcee, built as a 400B-parameter sparse Mixture-of-Experts with 13B active parameters per token using 4-of-256 expert routing. It excels in creative writing,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"structured_outputs",
"temperature",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "arcee-ai/trinity-large-preview:free",
"name": "Trinity Large Preview",
"provider": "openrouter",
"family": "trinity",
"created_at": "2026-01-28 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2025-06"
}
},
{
"id": "arcee-ai/trinity-large-thinking",
"name": "Trinity Large Thinking",
"provider": "openrouter",
"family": "trinity",
"created_at": "2026-04-01 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 80000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.22,
"output_per_million": 0.85
}
}
},
"metadata": {
"description": "Trinity Large Thinking is a powerful open source reasoning model from the team at Arcee AI. It shows strong performance in PinchBench, agentic workloads, and reasoning tasks. Launch video: https://youtu.be/Gc82AXLa0Rg?si=4RLn6WBz33qT--B7",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-03",
"cost": {
"input": 0.22,
"output": 0.85
},
"limit": {
"context": 262144,
"output": 80000
}
}
},
{
"id": "arcee-ai/trinity-mini",
"name": "Arcee AI: Trinity Mini",
"provider": "openrouter",
"family": "arcee-ai",
"created_at": "2025-12-01 15:08:40 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.045,
"output_per_million": 0.15
}
}
},
"metadata": {
"description": "Trinity Mini is a 26B-parameter (3B active) sparse mixture-of-experts language model featuring 128 experts with 8 active per token. Engineered for efficient reasoning over long contexts (131k) with robust function...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "arcee-ai/virtuoso-large",
"name": "Arcee AI: Virtuoso Large",
"provider": "openrouter",
"family": "arcee-ai",
"created_at": "2025-05-05 21:01:25 UTC",
"context_window": 131072,
"max_output_tokens": 64000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.75,
"output_per_million": 1.2
}
}
},
"metadata": {
"description": "Virtuoso‑Large is Arcee's top‑tier general‑purpose LLM at 72 B parameters, tuned to tackle cross‑domain reasoning, creative writing and enterprise QA. Unlike many 70 B peers, it retains the 128 k...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 64000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "baidu/cobuddy:free",
"name": "Baidu Qianfan: CoBuddy (free)",
"provider": "openrouter",
"family": "baidu",
"created_at": "2026-05-06 02:44:40 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"description": "CoBuddy is a code generation model from Baidu, optimized for coding tasks and AI Agent workflows. It features high inference throughput and low end-to-end latency, with native support for tool...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"stop",
"tools"
]
}
},
{
"id": "baidu/ernie-4.5-21b-a3b",
"name": "Baidu: ERNIE 4.5 21B A3B",
"provider": "openrouter",
"family": "baidu",
"created_at": "2025-08-12 21:29:27 UTC",
"context_window": 120000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.28
}
}
},
"metadata": {
"description": "A sophisticated text-based Mixture-of-Experts (MoE) model featuring 21B total parameters with 3B activated per token, delivering exceptional multimodal understanding and generation through heterogeneous MoE structures and modality-isolated routing. Supporting an...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 120000,
"max_completion_tokens": 8000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "baidu/ernie-4.5-21b-a3b-thinking",
"name": "Baidu: ERNIE 4.5 21B A3B Thinking",
"provider": "openrouter",
"family": "baidu",
"created_at": "2025-10-09 22:28:07 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.28
}
}
},
"metadata": {
"description": "ERNIE-4.5-21B-A3B-Thinking is Baidu's upgraded lightweight MoE model, refined to boost reasoning depth and quality for top-tier performance in logical puzzles, math, science, coding, text generation, and expert-level academic benchmarks.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "baidu/ernie-4.5-300b-a47b",
"name": "Baidu: ERNIE 4.5 300B A47B ",
"provider": "openrouter",
"family": "baidu",
"created_at": "2025-06-30 16:15:39 UTC",
"context_window": 123000,
"max_output_tokens": 12000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.28,
"output_per_million": 1.1
}
}
},
"metadata": {
"description": "ERNIE-4.5-300B-A47B is a 300B parameter Mixture-of-Experts (MoE) language model developed by Baidu as part of the ERNIE 4.5 series. It activates 47B parameters per token and supports text generation in...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 123000,
"max_completion_tokens": 12000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "baidu/ernie-4.5-vl-28b-a3b",
"name": "Baidu: ERNIE 4.5 VL 28B A3B",
"provider": "openrouter",
"family": "baidu",
"created_at": "2025-08-12 21:07:16 UTC",
"context_window": 30000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.56
}
}
},
"metadata": {
"description": "A powerful multimodal Mixture-of-Experts chat model featuring 28B total parameters with 3B activated per token, delivering exceptional text and vision understanding through its innovative heterogeneous MoE structure with modality-isolated routing....",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 30000,
"max_completion_tokens": 8000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "baidu/ernie-4.5-vl-424b-a47b",
"name": "Baidu: ERNIE 4.5 VL 424B A47B ",
"provider": "openrouter",
"family": "baidu",
"created_at": "2025-06-30 16:28:23 UTC",
"context_window": 123000,
"max_output_tokens": 16000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.42,
"output_per_million": 1.25
}
}
},
"metadata": {
"description": "ERNIE-4.5-VL-424B-A47B is a multimodal Mixture-of-Experts (MoE) model from Baidu’s ERNIE 4.5 series, featuring 424B total parameters with 47B active per token. It is trained jointly on text and image data...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 123000,
"max_completion_tokens": 16000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "baidu/qianfan-ocr-fast:free",
"name": "Baidu: Qianfan-OCR-Fast (free)",
"provider": "openrouter",
"family": "baidu",
"created_at": "2026-04-20 17:51:12 UTC",
"context_window": 65536,
"max_output_tokens": 28672,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"description": "Qianfan-OCR-Fast is a domain-specific multimodal large model purpose-built for OCR. By leveraging specialized OCR training data while preserving versatile multimodal intelligence, it provides a powerful performance upgrade over Qianfan-OCR.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 28672,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_p"
]
}
},
{
"id": "black-forest-labs/flux.2-flex",
"name": "FLUX.2 Flex",
"provider": "openrouter",
"family": "flux",
"created_at": "2025-11-25 00:00:00 UTC",
"context_window": 67344,
"max_output_tokens": 67344,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 67344,
"output": 67344
},
"knowledge": "2025-06"
}
},
{
"id": "black-forest-labs/flux.2-klein-4b",
"name": "FLUX.2 Klein 4B",
"provider": "openrouter",
"family": "flux",
"created_at": "2026-01-14 00:00:00 UTC",
"context_window": 40960,
"max_output_tokens": 40960,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 40960,
"output": 40960
},
"knowledge": "2025-06"
}
},
{
"id": "black-forest-labs/flux.2-max",
"name": "FLUX.2 Max",
"provider": "openrouter",
"family": "flux",
"created_at": "2025-12-16 00:00:00 UTC",
"context_window": 46864,
"max_output_tokens": 46864,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 46864,
"output": 46864
},
"knowledge": "2025-06"
}
},
{
"id": "black-forest-labs/flux.2-pro",
"name": "FLUX.2 Pro",
"provider": "openrouter",
"family": "flux",
"created_at": "2025-11-25 00:00:00 UTC",
"context_window": 46864,
"max_output_tokens": 46864,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 46864,
"output": 46864
},
"knowledge": "2025-06"
}
},
{
"id": "bytedance-seed/seed-1.6",
"name": "ByteDance Seed: Seed 1.6",
"provider": "openrouter",
"family": "bytedance-seed",
"created_at": "2025-12-23 15:49:57 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0
}
}
},
"metadata": {
"description": "Seed 1.6 is a general-purpose model released by the ByteDance Seed team. It incorporates multimodal capabilities and adaptive deep thinking with a 256K context window.",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "bytedance-seed/seed-1.6-flash",
"name": "ByteDance Seed: Seed 1.6 Flash",
"provider": "openrouter",
"family": "bytedance-seed",
"created_at": "2025-12-23 15:50:11 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"description": "Seed 1.6 Flash is an ultra-fast multimodal deep thinking model by ByteDance Seed, supporting both text and visual understanding. It features a 256k context window and can generate outputs of...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "bytedance-seed/seed-2.0-lite",
"name": "ByteDance Seed: Seed-2.0-Lite",
"provider": "openrouter",
"family": "bytedance-seed",
"created_at": "2026-03-10 15:40:31 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2.0
}
}
},
"metadata": {
"description": "Seed-2.0-Lite is a versatile, cost‑efficient enterprise workhorse that delivers strong multimodal and agent capabilities while offering noticeably lower latency, making it a practical default choice for most production workloads across...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "bytedance-seed/seed-2.0-mini",
"name": "ByteDance Seed: Seed-2.0-Mini",
"provider": "openrouter",
"family": "bytedance-seed",
"created_at": "2026-02-26 18:38:27 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.39999999999999997
}
}
},
"metadata": {
"description": "Seed-2.0-mini targets latency-sensitive, high-concurrency, and cost-sensitive scenarios, emphasizing fast response and flexible inference deployment. It delivers performance comparable to ByteDance-Seed-1.6, supports 256k context, four reasoning effort modes (minimal/low/medium/high), multimodal understanding,...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "bytedance-seed/seedream-4.5",
"name": "Seedream 4.5",
"provider": "openrouter",
"family": "seed",
"created_at": "2025-12-23 00:00:00 UTC",
"context_window": 4096,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 4096,
"output": 4096
},
"knowledge": "2025-06"
}
},
{
"id": "bytedance/ui-tars-1.5-7b",
"name": "ByteDance: UI-TARS 7B ",
"provider": "openrouter",
"family": "bytedance",
"created_at": "2025-07-22 17:24:16 UTC",
"context_window": 128000,
"max_output_tokens": 2048,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.19999999999999998,
"cache_read_input_per_million": 0.09999999999999999
}
}
},
"metadata": {
"description": "UI-TARS-1.5 is a multimodal vision-language agent optimized for GUI-based environments, including desktop interfaces, web browsers, mobile systems, and games. Built by ByteDance, it builds upon the UI-TARS framework with reinforcement...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 2048,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "cognitivecomputations/dolphin-mistral-24b-venice-edition:free",
"name": "Uncensored (free)",
"provider": "openrouter",
"family": "mistral",
"created_at": "2025-07-09 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 32768
},
"knowledge": "2025-06"
}
},
{
"id": "cohere/command-a",
"name": "Cohere: Command A",
"provider": "openrouter",
"family": "cohere",
"created_at": "2025-03-13 19:32:22 UTC",
"context_window": 256000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "Command A is an open-weights 111B parameter model with a 256k context window focused on delivering great performance across agentic, multilingual, and coding use cases. Compared to other leading proprietary...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 8192,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "cohere/command-r-08-2024",
"name": "Cohere: Command R (08-2024)",
"provider": "openrouter",
"family": "cohere",
"created_at": "2024-08-30 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "command-r-08-2024 is an update of the [Command R](/models/cohere/command-r) with improved performance for multilingual retrieval-augmented generation (RAG) and tool use. More broadly, it is better at math, code and reasoning and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Cohere",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "cohere/command-r-plus-08-2024",
"name": "Cohere: Command R+ (08-2024)",
"provider": "openrouter",
"family": "cohere",
"created_at": "2024-08-30 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "command-r-plus-08-2024 is an update of the [Command R+](/models/cohere/command-r-plus) with roughly 50% higher throughput and 25% lower latencies as compared to the previous Command R+ version, while keeping the hardware footprint...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Cohere",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "cohere/command-r7b-12-2024",
"name": "Cohere: Command R7B (12-2024)",
"provider": "openrouter",
"family": "cohere",
"created_at": "2024-12-14 06:35:52 UTC",
"context_window": 128000,
"max_output_tokens": 4000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.0375,
"output_per_million": 0.15
}
}
},
"metadata": {
"description": "Command R7B (12-2024) is a small, fast update of the Command R+ model, delivered in December 2024. It excels at RAG, tool use, agents, and similar tasks requiring complex reasoning...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Cohere",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "deepcogito/cogito-v2.1-671b",
"name": "Deep Cogito: Cogito v2.1 671B",
"provider": "openrouter",
"family": "deepcogito",
"created_at": "2025-11-13 22:00:33 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 1.25
}
}
},
"metadata": {
"description": "Cogito v2.1 671B MoE represents one of the strongest open models globally, matching performance of frontier closed and open models. This model is trained using self play with reinforcement learning...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "deepseek/deepseek-chat",
"name": "DeepSeek: DeepSeek V3",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2024-12-26 19:28:40 UTC",
"context_window": 163840,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.32,
"output_per_million": 0.8899999999999999
}
}
},
"metadata": {
"description": "DeepSeek-V3 is the latest model from the DeepSeek team, building upon the instruction following and coding abilities of the previous versions. Pre-trained on nearly 15 trillion tokens, the reported evaluations...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "deepseek/deepseek-chat-v3-0324",
"name": "DeepSeek V3 0324",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-03-24 00:00:00 UTC",
"context_window": 16384,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"streaming",
"function_calling",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.19999999999999998,
"output_per_million": 0.77,
"cache_read_input_per_million": 0.135
}
}
},
"metadata": {
"description": "DeepSeek V3, a 685B-parameter, mixture-of-experts model, is the latest iteration of the flagship chat model family from the DeepSeek team. It succeeds the [DeepSeek V3](/deepseek/deepseek-chat-v3) model and performs really well...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-03-24",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 16384,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "deepseek/deepseek-chat-v3.1",
"name": "DeepSeek-V3.1",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-08-21 00:00:00 UTC",
"context_window": 163840,
"max_output_tokens": 163840,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.8
}
}
},
"metadata": {
"description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": "deepseek-v3.1"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 7168,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-21",
"cost": {
"input": 0.2,
"output": 0.8
},
"limit": {
"context": 163840,
"output": 163840
},
"knowledge": "2025-07"
}
},
{
"id": "deepseek/deepseek-r1",
"name": "DeepSeek: R1",
"provider": "openrouter",
"family": "deepseek-thinking",
"created_at": "2025-01-20 00:00:00 UTC",
"context_window": 64000,
"max_output_tokens": 16000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7,
"output_per_million": 2.5
}
}
},
"metadata": {
"description": "DeepSeek R1 is here: Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": "deepseek-r1"
},
"top_provider": {
"context_length": 64000,
"max_completion_tokens": 16000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-01-20",
"cost": {
"input": 0.7,
"output": 2.5
},
"limit": {
"context": 64000,
"output": 16000
},
"knowledge": "2024-07"
}
},
{
"id": "deepseek/deepseek-r1-0528",
"name": "DeepSeek: R1 0528",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-05-28 17:59:30 UTC",
"context_window": 163840,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 2.1500000000000004,
"cache_read_input_per_million": 0.35
}
}
},
"metadata": {
"description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": "deepseek-r1"
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "deepseek/deepseek-r1-distill-llama-70b",
"name": "DeepSeek R1 Distill Llama 70B",
"provider": "openrouter",
"family": "deepseek-thinking",
"created_at": "2025-01-23 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7,
"output_per_million": 0.7999999999999999
}
}
},
"metadata": {
"description": "DeepSeek R1 Distill Llama 70B is a distilled large language model based on [Llama-3.3-70B-Instruct](/meta-llama/llama-3.3-70b-instruct), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). The model combines advanced distillation techniques to achieve high performance across...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "deepseek-r1"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-01-23",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "deepseek/deepseek-r1-distill-qwen-32b",
"name": "DeepSeek: R1 Distill Qwen 32B",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-01-29 23:53:50 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.29,
"output_per_million": 0.29
}
}
},
"metadata": {
"description": "DeepSeek R1 Distill Qwen 32B is a distilled large language model based on [Qwen 2.5 32B](https://huggingface.co/Qwen/Qwen2.5-32B), using outputs from [DeepSeek R1](/deepseek/deepseek-r1). It outperforms OpenAI's o1-mini across various benchmarks, achieving new...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": "deepseek-r1"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_logprobs",
"top_p"
]
}
},
{
"id": "deepseek/deepseek-v3.1-terminus",
"name": "DeepSeek V3.1 Terminus",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-09-22 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.27,
"output_per_million": 1
}
}
},
"metadata": {
"description": "DeepSeek-V3.1 Terminus is an update to [DeepSeek V3.1](/deepseek/deepseek-chat-v3.1) that maintains the model's original capabilities while addressing issues reported by users, including language consistency and agent capabilities, further optimizing the model's...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": "deepseek-v3.1"
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-22",
"cost": {
"input": 0.27,
"output": 1
},
"limit": {
"context": 131072,
"output": 65536
},
"knowledge": "2025-07"
}
},
{
"id": "deepseek/deepseek-v3.1-terminus:exacto",
"name": "DeepSeek V3.1 Terminus (exacto)",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-09-22 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.27,
"output_per_million": 1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-22",
"cost": {
"input": 0.27,
"output": 1
},
"limit": {
"context": 131072,
"output": 65536
},
"knowledge": "2025-07"
}
},
{
"id": "deepseek/deepseek-v3.2",
"name": "DeepSeek V3.2",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-12-01 00:00:00 UTC",
"context_window": 163840,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.28,
"output_per_million": 0.4
}
}
},
"metadata": {
"description": "DeepSeek-V3.2 is a large language model designed to harmonize high computational efficiency with strong reasoning and agentic tool-use performance. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-01",
"cost": {
"input": 0.28,
"output": 0.4
},
"limit": {
"context": 163840,
"output": 65536
},
"knowledge": "2024-07"
}
},
{
"id": "deepseek/deepseek-v3.2-exp",
"name": "DeepSeek: DeepSeek V3.2 Exp",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-09-29 12:54:41 UTC",
"context_window": 163840,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.27,
"output_per_million": 0.41
}
}
},
"metadata": {
"description": "DeepSeek-V3.2-Exp is an experimental large language model released by DeepSeek as an intermediate step between V3.1 and future architectures. It introduces DeepSeek Sparse Attention (DSA), a fine-grained sparse attention mechanism...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": "deepseek-v3.1"
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "deepseek/deepseek-v3.2-speciale",
"name": "DeepSeek V3.2 Speciale",
"provider": "openrouter",
"family": "deepseek",
"created_at": "2025-12-01 00:00:00 UTC",
"context_window": 163840,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.27,
"output_per_million": 0.41
}
}
},
"metadata": {
"description": "DeepSeek-V3.2-Speciale is a high-compute variant of DeepSeek-V3.2 optimized for maximum reasoning and agentic performance. It builds on DeepSeek Sparse Attention (DSA) for efficient long-context processing, then scales post-training reinforcement learning...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 163840,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-01",
"cost": {
"input": 0.27,
"output": 0.41
},
"limit": {
"context": 163840,
"output": 65536
},
"knowledge": "2024-07"
}
},
{
"id": "deepseek/deepseek-v4-flash",
"name": "DeepSeek V4 Flash",
"provider": "openrouter",
"family": "deepseek-flash",
"created_at": "2026-04-24 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 393216,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.28,
"cache_read_input_per_million": 0.028
}
}
},
"metadata": {
"description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 384000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-24",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0.14,
"output": 0.28,
"cache_read": 0.028
},
"limit": {
"context": 1048576,
"output": 393216
},
"knowledge": "2025-05"
}
},
{
"id": "deepseek/deepseek-v4-pro",
"name": "DeepSeek V4 Pro",
"provider": "openrouter",
"family": "deepseek-thinking",
"created_at": "2026-04-24 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 393216,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.74,
"output_per_million": 3.48,
"cache_read_input_per_million": 0.145
}
}
},
"metadata": {
"description": "DeepSeek V4 Pro is a large-scale Mixture-of-Experts model from DeepSeek with 1.6T total parameters and 49B activated parameters, supporting a 1M-token context window. It is designed for advanced reasoning, coding,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 384000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-24",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 1.74,
"output": 3.48,
"cache_read": 0.145
},
"limit": {
"context": 1048576,
"output": 393216
},
"knowledge": "2025-05"
}
},
{
"id": "essentialai/rnj-1-instruct",
"name": "EssentialAI: Rnj 1 Instruct",
"provider": "openrouter",
"family": "essentialai",
"created_at": "2025-12-07 08:07:27 UTC",
"context_window": 32768,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15
}
}
},
"metadata": {
"description": "Rnj-1 is an 8B-parameter, dense, open-weight model family developed by Essential AI and trained from scratch with a focus on programming, math, and scientific reasoning. The model demonstrates strong performance...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "google/gemini-2.0-flash-001",
"name": "Gemini 2.0 Flash",
"provider": "openrouter",
"family": "gemini-flash",
"created_at": "2024-12-11 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"description": "Gemini Flash 2.0 offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5). It...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-11",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 8192
},
"knowledge": "2024-06"
}
},
{
"id": "google/gemini-2.0-flash-lite-001",
"name": "Google: Gemini 2.0 Flash Lite",
"provider": "openrouter",
"family": "google",
"created_at": "2025-02-25 17:56:52 UTC",
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3,
"reasoning_output_per_million": 0.3
}
}
},
"metadata": {
"description": "Gemini 2.0 Flash Lite offers a significantly faster time to first token (TTFT) compared to [Gemini Flash 1.5](/google/gemini-flash-1.5), while maintaining quality on par with larger models like [Gemini Pro 1.5](/google/gemini-pro-1.5),...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "google/gemini-2.5-flash",
"name": "Gemini 2.5 Flash",
"provider": "openrouter",
"family": "gemini-flash",
"created_at": "2025-07-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.0375
}
}
},
"metadata": {
"description": "Gemini 2.5 Flash is Google's state-of-the-art workhorse model, specifically designed for advanced reasoning, coding, mathematics, and scientific tasks. It includes built-in \"thinking\" capabilities, enabling it to provide responses with greater...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"file",
"image",
"text",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65535,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-07-17",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.0375
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-2.5-flash-image",
"name": "Google: Nano Banana (Gemini 2.5 Flash Image)",
"provider": "openrouter",
"family": "google",
"created_at": "2025-10-07 20:53:51 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image",
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.03,
"reasoning_output_per_million": 2.5
}
}
},
"metadata": {
"description": "Gemini 2.5 Flash Image, a.k.a. \"Nano Banana,\" is now generally available. It is a state of the art image generation model with contextual understanding. It is capable of image generation,...",
"architecture": {
"modality": "text+image->text+image",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"image",
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_p"
]
}
},
{
"id": "google/gemini-2.5-flash-lite",
"name": "Gemini 2.5 Flash Lite",
"provider": "openrouter",
"family": "gemini-flash-lite",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65535,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-2.5-flash-lite-preview-09-2025",
"name": "Gemini 2.5 Flash Lite Preview 09-25",
"provider": "openrouter",
"family": "gemini-flash-lite",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"description": "Gemini 2.5 Flash-Lite is a lightweight reasoning model in the Gemini 2.5 family, optimized for ultra-low latency and cost efficiency. It offers improved throughput, faster token generation, and better performance...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65535,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-2.5-flash-preview-09-2025",
"name": "Gemini 2.5 Flash Preview 09-25",
"provider": "openrouter",
"family": "gemini-flash",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.031
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.031
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-2.5-pro",
"name": "Gemini 2.5 Pro",
"provider": "openrouter",
"family": "gemini-pro",
"created_at": "2025-03-20 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125,
"context_over_200k": {
"input": 2.5,
"output": 15,
"cache_read": 0.25
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-2.5-pro-preview",
"name": "Google: Gemini 2.5 Pro Preview 06-05",
"provider": "openrouter",
"family": "google",
"created_at": "2025-06-05 15:27:37 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"file",
"image",
"text",
"audio"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10.0,
"cache_read_input_per_million": 0.125,
"reasoning_output_per_million": 10.0
}
}
},
"metadata": {
"description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...",
"architecture": {
"modality": "text+image+file+audio->text",
"input_modalities": [
"file",
"image",
"text",
"audio"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "google/gemini-2.5-pro-preview-05-06",
"name": "Gemini 2.5 Pro Preview 05-06",
"provider": "openrouter",
"family": "gemini-pro",
"created_at": "2025-05-06 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.31
}
}
},
"metadata": {
"description": "Gemini 2.5 Pro is Google’s state-of-the-art AI model designed for advanced reasoning, coding, mathematics, and scientific tasks. It employs “thinking” capabilities, enabling it to reason through responses with enhanced accuracy...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65535,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-06",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.31
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-2.5-pro-preview-06-05",
"name": "Gemini 2.5 Pro Preview 06-05",
"provider": "openrouter",
"family": "gemini-pro",
"created_at": "2025-06-05 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.31
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.31
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-3-flash-preview",
"name": "Gemini 3 Flash Preview",
"provider": "openrouter",
"family": "gemini-flash",
"created_at": "2025-12-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 3,
"cache_read_input_per_million": 0.05
}
}
},
"metadata": {
"description": "Gemini 3 Flash Preview is a high speed, high value thinking model designed for agentic workflows, multi turn chat, and coding assistance. It delivers near Pro level reasoning and tool...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-12-17",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.5,
"output": 3,
"cache_read": 0.05
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-3-pro-image-preview",
"name": "Google: Nano Banana Pro (Gemini 3 Pro Image Preview)",
"provider": "openrouter",
"family": "google",
"created_at": "2025-11-20 15:49:57 UTC",
"context_window": 65536,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"image",
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 12.0,
"cache_read_input_per_million": 0.19999999999999998,
"reasoning_output_per_million": 12.0
}
}
},
"metadata": {
"description": "Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...",
"architecture": {
"modality": "text+image->text+image",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"image",
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_p"
]
}
},
{
"id": "google/gemini-3-pro-preview",
"name": "Gemini 3 Pro Preview",
"provider": "openrouter",
"family": "gemini-pro",
"created_at": "2025-11-18 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 66000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 2,
"output": 12
},
"limit": {
"context": 1050000,
"output": 66000
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-3.1-flash-image-preview",
"name": "Gemini 3.1 Flash Image Preview (Nano Banana 2)",
"provider": "openrouter",
"family": "gemini-flash",
"created_at": "2026-02-26 00:00:00 UTC",
"context_window": 65536,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 3
}
}
},
"metadata": {
"description": "Gemini 3.1 Flash Image Preview, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines...",
"architecture": {
"modality": "text+image->text+image",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"image",
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-26",
"cost": {
"input": 0.5,
"output": 3
},
"limit": {
"context": 65536,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-3.1-flash-lite-preview",
"name": "Gemini 3.1 Flash Lite Preview",
"provider": "openrouter",
"family": "gemini-flash-lite",
"created_at": "2026-03-03 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"pdf",
"audio"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.5,
"cache_read_input_per_million": 0.025,
"cache_write_input_per_million": 0.083,
"reasoning_output_per_million": 1.5
}
},
"audio_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 0.5
}
}
},
"metadata": {
"description": "Gemini 3.1 Flash Lite Preview is Google's high-efficiency model optimized for high-volume use cases. It outperforms Gemini 2.5 Flash Lite on overall quality and approaches Gemini 2.5 Flash performance across...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"video",
"file",
"audio"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-03",
"cost": {
"input": 0.25,
"output": 1.5,
"reasoning": 1.5,
"cache_read": 0.025,
"cache_write": 0.083,
"input_audio": 0.5,
"output_audio": 0.5
},
"limit": {
"context": 1048576,
"output": 65536
}
}
},
{
"id": "google/gemini-3.1-pro-preview",
"name": "Gemini 3.1 Pro Preview",
"provider": "openrouter",
"family": "gemini-pro",
"created_at": "2026-02-19 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"reasoning_output_per_million": 12
}
}
},
"metadata": {
"description": "Gemini 3.1 Pro Preview is Google’s frontier reasoning model, delivering enhanced software engineering performance, improved agentic reliability, and more efficient token usage across complex workflows. Building on the multimodal foundation...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"audio",
"file",
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-19",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 2,
"output": 12,
"reasoning": 12,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemini-3.1-pro-preview-customtools",
"name": "Gemini 3.1 Pro Preview Custom Tools",
"provider": "openrouter",
"family": "gemini-pro",
"created_at": "2026-02-19 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"reasoning_output_per_million": 12
}
}
},
"metadata": {
"description": "Gemini 3.1 Pro Preview Custom Tools is a variant of Gemini 3.1 Pro that improves tool selection behavior by preventing overuse of a general bash tool when more efficient third-party...",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"audio",
"image",
"video",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-19",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 2,
"output": 12,
"reasoning": 12,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemma-2-27b-it",
"name": "Google: Gemma 2 27B",
"provider": "openrouter",
"family": "google",
"created_at": "2024-07-13 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 2048,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.65,
"output_per_million": 0.65
}
}
},
"metadata": {
"description": "Gemma 2 27B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini). Gemma models are well-suited for a variety of...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": "gemma"
},
"top_provider": {
"context_length": 8192,
"max_completion_tokens": 2048,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_p"
]
}
},
{
"id": "google/gemma-2-9b-it",
"name": "Gemma 2 9B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2024-06-28 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.09
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-06-28",
"cost": {
"input": 0.03,
"output": 0.09
},
"limit": {
"context": 8192,
"output": 8192
},
"knowledge": "2024-06"
}
},
{
"id": "google/gemma-3-12b-it",
"name": "Gemma 3 12B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-03-13 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"vision",
"streaming",
"function_calling",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.1
}
}
},
"metadata": {
"description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": "gemma"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-13",
"cost": {
"input": 0.03,
"output": 0.1
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2024-10"
}
},
{
"id": "google/gemma-3-12b-it:free",
"name": "Gemma 3 12B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-03-13 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-13",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "google/gemma-3-27b-it",
"name": "Gemma 3 27B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-03-12 00:00:00 UTC",
"context_window": 96000,
"max_output_tokens": 96000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.15
}
}
},
"metadata": {
"description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": "gemma"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-12",
"cost": {
"input": 0.04,
"output": 0.15
},
"limit": {
"context": 96000,
"output": 96000
},
"knowledge": "2024-10"
}
},
{
"id": "google/gemma-3-27b-it:free",
"name": "Gemma 3 27B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-03-12 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-12",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "google/gemma-3-4b-it",
"name": "Gemma 3 4B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-03-13 00:00:00 UTC",
"context_window": 96000,
"max_output_tokens": 96000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.01703,
"output_per_million": 0.06815
}
}
},
"metadata": {
"description": "Gemma 3 introduces multimodality, supporting vision-language input and text outputs. It handles context windows up to 128k tokens, understands over 140 languages, and offers improved math, reasoning, and chat capabilities,...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemini",
"instruct_type": "gemma"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-13",
"cost": {
"input": 0.01703,
"output": 0.06815
},
"limit": {
"context": 96000,
"output": 96000
},
"knowledge": "2024-10"
}
},
{
"id": "google/gemma-3-4b-it:free",
"name": "Gemma 3 4B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-03-13 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-13",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "google/gemma-3n-e2b-it:free",
"name": "Gemma 3n 2B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-07-09 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-07-09",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 2000
},
"knowledge": "2024-06"
}
},
{
"id": "google/gemma-3n-e4b-it",
"name": "Gemma 3n 4B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.02,
"output_per_million": 0.04
}
}
},
"metadata": {
"description": "Gemma 3n E4B-it is optimized for efficient execution on mobile and low-resource devices, such as phones, laptops, and tablets. It supports multimodal inputs—including text, visual data, and audio—enabling diverse tasks...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-20",
"cost": {
"input": 0.02,
"output": 0.04
},
"limit": {
"context": 32768,
"output": 32768
},
"knowledge": "2024-06"
}
},
{
"id": "google/gemma-3n-e4b-it:free",
"name": "Gemma 3n 4B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-20",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 2000
},
"knowledge": "2024-06"
}
},
{
"id": "google/gemma-4-26b-a4b-it",
"name": "Gemma 4 26B A4B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2026-04-03 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.13,
"output_per_million": 0.4
}
}
},
"metadata": {
"description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemma",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-03",
"cost": {
"input": 0.13,
"output": 0.4
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemma-4-26b-a4b-it:free",
"name": "Gemma 4 26B A4B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2026-04-03 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Gemma 4 26B A4B IT is an instruction-tuned Mixture-of-Experts (MoE) model from Google DeepMind. Despite 25.2B total parameters, only 3.8B activate per token during inference — delivering near-31B quality at...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemma",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-03",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 262144,
"output": 32768
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemma-4-31b-it",
"name": "Gemma 4 31B",
"provider": "openrouter",
"family": "gemma",
"created_at": "2026-04-02 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.4
}
}
},
"metadata": {
"description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemma",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-02",
"cost": {
"input": 0.14,
"output": 0.4
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-01"
}
},
{
"id": "google/gemma-4-31b-it:free",
"name": "Gemma 4 31B (free)",
"provider": "openrouter",
"family": "gemma",
"created_at": "2026-04-02 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Gemma 4 31B Instruct is Google DeepMind's 30.7B dense multimodal model supporting text and image input with text output. Features a 256K token context window, configurable thinking/reasoning mode, native function...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Gemma",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-02",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 262144,
"output": 32768
},
"knowledge": "2025-01"
}
},
{
"id": "google/lyria-3-clip-preview",
"name": "Google: Lyria 3 Clip Preview",
"provider": "openrouter",
"family": "google",
"created_at": "2026-03-30 21:47:35 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {},
"metadata": {
"description": "30 second duration clips are priced at $0.04 per clip. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate...",
"architecture": {
"modality": "text+image->text+audio",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text",
"audio"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"seed",
"temperature",
"top_p"
]
}
},
{
"id": "google/lyria-3-pro-preview",
"name": "Google: Lyria 3 Pro Preview",
"provider": "openrouter",
"family": "google",
"created_at": "2026-03-30 21:48:06 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {},
"metadata": {
"description": "Full-length songs are priced at $0.08 per song. Lyria 3 is Google's family of music generation models, available through the Gemini API. With Lyria 3, you can generate high-quality, 48kHz...",
"architecture": {
"modality": "text+image->text+audio",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text",
"audio"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"seed",
"temperature",
"top_p"
]
}
},
{
"id": "gryphe/mythomax-l2-13b",
"name": "MythoMax 13B",
"provider": "openrouter",
"family": "gryphe",
"created_at": "2023-07-02 00:00:00 UTC",
"context_window": 4096,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.06
}
}
},
"metadata": {
"description": "One of the highest performing and most popular fine-tunes of Llama 2 13B, with rich descriptions and roleplay. #merge",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama2",
"instruct_type": "alpaca"
},
"top_provider": {
"context_length": 4096,
"max_completion_tokens": 4096,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_a",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "ibm-granite/granite-4.0-h-micro",
"name": "IBM: Granite 4.0 Micro",
"provider": "openrouter",
"family": "ibm-granite",
"created_at": "2025-10-20 02:34:55 UTC",
"context_window": 131000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.017,
"output_per_million": 0.11
}
}
},
"metadata": {
"description": "Granite-4.0-H-Micro is a 3B parameter from the Granite 4 family of models. These models are the latest in a series of models released by IBM. They are fine-tuned for long...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "ibm-granite/granite-4.1-8b",
"name": "IBM: Granite 4.1 8B",
"provider": "openrouter",
"family": "ibm-granite",
"created_at": "2026-04-30 19:24:31 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.049999999999999996,
"output_per_million": 0.09999999999999999,
"cache_read_input_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "Granite 4.1 8B is a dense, decoder-only 8-billion-parameter language model from IBM, part of the Granite 4.1 family. It supports a 131K-token context window and is designed for enterprise tasks...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "inception/mercury-2",
"name": "Mercury 2",
"provider": "openrouter",
"family": "mercury",
"created_at": "2026-03-04 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 50000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 0.75,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"description": "Mercury 2 is an extremely fast reasoning LLM, and the first reasoning diffusion LLM (dLLM). Instead of generating tokens sequentially, Mercury 2 produces and refines multiple tokens in parallel, achieving...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 50000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-04",
"cost": {
"input": 0.25,
"output": 0.75,
"cache_read": 0.025
},
"limit": {
"context": 128000,
"output": 50000
}
}
},
{
"id": "inception/mercury-edit-2",
"name": "Mercury Edit 2",
"provider": "openrouter",
"family": null,
"created_at": "2026-03-30 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 0.75,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-30",
"cost": {
"input": 0.25,
"output": 0.75,
"cache_read": 0.025
},
"limit": {
"context": 128000,
"output": 8192
}
}
},
{
"id": "inclusionai/ling-2.6-1t:free",
"name": "inclusionAI: Ling-2.6-1T (free)",
"provider": "openrouter",
"family": "inclusionai",
"created_at": "2026-04-23 12:43:58 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"description": "Ling-2.6-1T is an instant (instruct) model from inclusionAI and the company’s trillion-parameter flagship, designed for real-world agents that require fast execution and high efficiency at scale. It uses a “fast...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "inclusionai/ling-2.6-flash",
"name": "inclusionAI: Ling-2.6-flash",
"provider": "openrouter",
"family": "inclusionai",
"created_at": "2026-04-21 18:24:46 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.08,
"output_per_million": 0.24,
"cache_read_input_per_million": 0.016
}
}
},
"metadata": {
"description": "Ling-2.6-flash is an instant (instruct) model from inclusionAI with 104B total parameters and 7.4B active parameters, designed for real-world agents that require fast responses, strong execution, and high token efficiency....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "inflection/inflection-3-pi",
"name": "Inflection: Inflection 3 Pi",
"provider": "openrouter",
"family": "inflection",
"created_at": "2024-10-11 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 1024,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "Inflection 3 Pi powers Inflection's [Pi](https://pi.ai) chatbot, including backstory, emotional intelligence, productivity, and safety. It has access to recent news, and excels in scenarios like customer support and roleplay. Pi...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 8000,
"max_completion_tokens": 1024,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"top_p"
]
}
},
{
"id": "inflection/inflection-3-productivity",
"name": "Inflection: Inflection 3 Productivity",
"provider": "openrouter",
"family": "inflection",
"created_at": "2024-10-11 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 1024,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "Inflection 3 Productivity is optimized for following instructions. It is better for tasks requiring JSON output or precise adherence to provided guidelines. It has access to recent news. For emotional...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 8000,
"max_completion_tokens": 1024,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"top_p"
]
}
},
{
"id": "kwaipilot/kat-coder-pro-v2",
"name": "Kwaipilot: KAT-Coder-Pro V2",
"provider": "openrouter",
"family": "kwaipilot",
"created_at": "2026-03-27 22:08:30 UTC",
"context_window": 256000,
"max_output_tokens": 80000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2,
"cache_read_input_per_million": 0.06
}
}
},
"metadata": {
"description": "KAT-Coder-Pro V2 is the latest high-performance model in KwaiKAT’s KAT-Coder series, designed for complex enterprise-grade software engineering and SaaS integration. It builds on the agentic coding strengths of earlier versions,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 80000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "liquid/lfm-2-24b-a2b",
"name": "LiquidAI: LFM2-24B-A2B",
"provider": "openrouter",
"family": "liquid",
"created_at": "2026-02-25 19:45:11 UTC",
"context_window": 32768,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.12
}
}
},
"metadata": {
"description": "LFM2-24B-A2B is the largest model in the LFM2 family of hybrid architectures designed for efficient on-device deployment. Built as a 24B parameter Mixture-of-Experts model with only 2B active parameters per...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "liquid/lfm-2.5-1.2b-instruct:free",
"name": "LFM2.5-1.2B-Instruct (free)",
"provider": "openrouter",
"family": "liquid",
"created_at": "2026-01-20 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"description": "LFM2.5-1.2B-Instruct is a compact, high-performance instruction-tuned model built for fast on-device AI. It delivers strong chat quality in a 1.2B parameter footprint, with efficient edge inference and broad runtime support.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 32768
},
"knowledge": "2025-06"
}
},
{
"id": "liquid/lfm-2.5-1.2b-thinking:free",
"name": "LFM2.5-1.2B-Thinking (free)",
"provider": "openrouter",
"family": "liquid",
"created_at": "2026-01-20 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "LFM2.5-1.2B-Thinking is a lightweight reasoning-focused model optimized for agentic tasks, data extraction, and RAG—while still running comfortably on edge devices. It supports long context (up to 32K tokens) and is...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 32768
},
"knowledge": "2025-06"
}
},
{
"id": "mancer/weaver",
"name": "Mancer: Weaver (alpha)",
"provider": "openrouter",
"family": "mancer",
"created_at": "2023-08-02 00:00:00 UTC",
"context_window": 8000,
"max_output_tokens": 2000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.75,
"output_per_million": 1.0
}
}
},
"metadata": {
"description": "An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama2",
"instruct_type": "alpaca"
},
"top_provider": {
"context_length": 8000,
"max_completion_tokens": 2000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_a",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3-70b-instruct",
"name": "Meta: Llama 3 70B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-04-18 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.51,
"output_per_million": 0.74
}
}
},
"metadata": {
"description": "Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 70B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 8192,
"max_completion_tokens": 8000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3-8b-instruct",
"name": "Meta: Llama 3 8B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-04-18 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.03,
"output_per_million": 0.04
}
}
},
"metadata": {
"description": "Meta's latest class of model (Llama 3) launched with a variety of sizes & flavors. This 8B instruct-tuned version was optimized for high quality dialogue usecases. It has demonstrated strong...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 8192,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3.1-70b-instruct",
"name": "Meta: Llama 3.1 70B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.39999999999999997,
"output_per_million": 0.39999999999999997
}
}
},
"metadata": {
"description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 70B instruct-tuned version is optimized for high quality dialogue usecases. It has demonstrated strong...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3.1-8b-instruct",
"name": "Meta: Llama 3.1 8B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-07-23 00:00:00 UTC",
"context_window": 16384,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.02,
"output_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "Meta's latest class of model (Llama 3.1) launched with a variety of sizes & flavors. This 8B instruct-tuned version is fast and efficient. It has demonstrated strong performance compared to...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 16384,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3.2-11b-vision-instruct",
"name": "Llama 3.2 11B Vision Instruct",
"provider": "openrouter",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.245,
"output_per_million": 0.245
}
}
},
"metadata": {
"description": "Llama 3.2 11B Vision is a multimodal model with 11 billion parameters, designed to handle tasks combining visual and textual data. It excels in tasks such as image captioning and...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2023-12"
}
},
{
"id": "meta-llama/llama-3.2-1b-instruct",
"name": "Meta: Llama 3.2 1B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 60000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.027,
"output_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Llama 3.2 1B is a 1-billion-parameter language model focused on efficiently performing natural language tasks, such as summarization, dialogue, and multilingual text analysis. Its smaller size allows it to operate...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 60000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3.2-3b-instruct",
"name": "Meta: Llama 3.2 3B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 80000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.051,
"output_per_million": 0.33999999999999997
}
}
},
"metadata": {
"description": "Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 80000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3.2-3b-instruct:free",
"name": "Llama 3.2 3B Instruct (free)",
"provider": "openrouter",
"family": "llama",
"created_at": "2024-09-25 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Llama 3.2 3B is a 3-billion-parameter multilingual large language model, optimized for advanced natural language processing tasks like dialogue generation, reasoning, and summarization. Designed with the latest transformer architecture, it...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2024-09-25",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2023-12"
}
},
{
"id": "meta-llama/llama-3.3-70b-instruct",
"name": "Meta: Llama 3.3 70B Instruct",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2024-12-06 17:28:57 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.32
}
}
},
"metadata": {
"description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-3.3-70b-instruct:free",
"name": "Llama 3.3 70B Instruct (free)",
"provider": "openrouter",
"family": "llama",
"created_at": "2024-12-06 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {},
"metadata": {
"description": "The Meta Llama 3.3 multilingual large language model (LLM) is a pretrained and instruction tuned generative model in 70B (text in/text out). The Llama 3.3 instruction tuned text only model...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-12-06",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2024-12"
}
},
{
"id": "meta-llama/llama-4-maverick",
"name": "Meta: Llama 4 Maverick",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2025-04-05 19:37:02 UTC",
"context_window": 1048576,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "Llama 4 Maverick 17B Instruct (128E) is a high-capacity multimodal language model from Meta, built on a mixture-of-experts (MoE) architecture with 128 experts and 17 billion active parameters per forward...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama4",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-4-scout",
"name": "Meta: Llama 4 Scout",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2025-04-05 19:31:59 UTC",
"context_window": 327680,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.08,
"output_per_million": 0.3
}
}
},
"metadata": {
"description": "Llama 4 Scout 17B Instruct (16E) is a mixture-of-experts (MoE) language model developed by Meta, activating 17 billion parameters out of a total of 109B. It supports native multimodal input...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama4",
"instruct_type": null
},
"top_provider": {
"context_length": 327680,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-guard-3-8b",
"name": "Llama Guard 3 8B",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2025-02-12 23:01:58 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.48,
"output_per_million": 0.03
}
}
},
"metadata": {
"description": "Llama Guard 3 is a Llama-3.1-8B pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM inputs (prompt classification)...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "none"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "meta-llama/llama-guard-4-12b",
"name": "Meta: Llama Guard 4 12B",
"provider": "openrouter",
"family": "meta-llama",
"created_at": "2025-04-30 01:06:33 UTC",
"context_window": 163840,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.18,
"output_per_million": 0.18
}
}
},
"metadata": {
"description": "Llama Guard 4 is a Llama 4 Scout-derived multimodal pretrained model, fine-tuned for content safety classification. Similar to previous versions, it can be used to classify content in both LLM...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "microsoft/phi-4",
"name": "Microsoft: Phi 4",
"provider": "openrouter",
"family": "microsoft",
"created_at": "2025-01-10 06:17:52 UTC",
"context_window": 16384,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.065,
"output_per_million": 0.14
}
}
},
"metadata": {
"description": "[Microsoft Research](/microsoft) Phi-4 is designed to perform well in complex reasoning tasks and can operate efficiently in situations with limited memory or where quick responses are needed. At 14 billion...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 16384,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "microsoft/phi-4-mini-instruct",
"name": "Microsoft: Phi 4 Mini Instruct",
"provider": "openrouter",
"family": "microsoft",
"created_at": "2025-10-17 18:34:09 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.08,
"output_per_million": 0.35,
"cache_read_input_per_million": 0.08
}
}
},
"metadata": {
"description": "Phi-4-mini-instruct is a lightweight open model built upon synthetic data and filtered publicly available websites - with a focus on high-quality, reasoning dense data. The model belongs to the Phi-4...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "microsoft/wizardlm-2-8x22b",
"name": "WizardLM-2 8x22B",
"provider": "openrouter",
"family": "microsoft",
"created_at": "2024-04-16 00:00:00 UTC",
"context_window": 65535,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.62,
"output_per_million": 0.62
}
}
},
"metadata": {
"description": "WizardLM-2 8x22B is Microsoft AI's most advanced Wizard model. It demonstrates highly competitive performance compared to leading proprietary models, and it consistently outperforms all existing state-of-the-art opensource models. It is...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": "vicuna"
},
"top_provider": {
"context_length": 65535,
"max_completion_tokens": 8000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "minimax/minimax-01",
"name": "MiniMax-01",
"provider": "openrouter",
"family": "minimax",
"created_at": "2025-01-15 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 1000000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 1.1
}
}
},
"metadata": {
"description": "MiniMax-01 is a combines MiniMax-Text-01 for text generation and MiniMax-VL-01 for image understanding. It has 456 billion parameters, with 45.9 billion parameters activated per inference, and can handle a context...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1000192,
"max_completion_tokens": 1000192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"temperature",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-01-15",
"cost": {
"input": 0.2,
"output": 1.1
},
"limit": {
"context": 1000000,
"output": 1000000
}
}
},
{
"id": "minimax/minimax-m1",
"name": "MiniMax M1",
"provider": "openrouter",
"family": "minimax",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 40000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2.2
}
}
},
"metadata": {
"description": "MiniMax-M1 is a large-scale, open-weight reasoning model designed for extended context and high-efficiency inference. It leverages a hybrid Mixture-of-Experts (MoE) architecture paired with a custom \"lightning attention\" mechanism, allowing it...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 40000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.4,
"output": 2.2
},
"limit": {
"context": 1000000,
"output": 40000
}
}
},
{
"id": "minimax/minimax-m2",
"name": "MiniMax M2",
"provider": "openrouter",
"family": "minimax",
"created_at": "2025-10-23 00:00:00 UTC",
"context_window": 196600,
"max_output_tokens": 118000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.28,
"output_per_million": 1.15,
"cache_read_input_per_million": 0.28,
"cache_write_input_per_million": 1.15
}
}
},
"metadata": {
"description": "MiniMax-M2 is a compact, high-efficiency large language model optimized for end-to-end coding and agentic workflows. With 10 billion activated parameters (230 billion total), it delivers near-frontier intelligence across general reasoning,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 196608,
"max_completion_tokens": 196608,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-10-23",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.28,
"output": 1.15,
"cache_read": 0.28,
"cache_write": 1.15
},
"limit": {
"context": 196600,
"output": 118000
}
}
},
{
"id": "minimax/minimax-m2-her",
"name": "MiniMax: MiniMax M2-her",
"provider": "openrouter",
"family": "minimax",
"created_at": "2026-01-23 14:07:19 UTC",
"context_window": 65536,
"max_output_tokens": 2048,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2,
"cache_read_input_per_million": 0.03
}
}
},
"metadata": {
"description": "MiniMax M2-her is a dialogue-first large language model built for immersive roleplay, character-driven chat, and expressive multi-turn conversations. Designed to stay consistent in tone and personality, it supports rich message...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 2048,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"temperature",
"top_p"
]
}
},
{
"id": "minimax/minimax-m2.1",
"name": "MiniMax M2.1",
"provider": "openrouter",
"family": "minimax",
"created_at": "2025-12-23 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2
}
}
},
"metadata": {
"description": "MiniMax-M2.1 is a lightweight, state-of-the-art large language model optimized for coding, agentic workflows, and modern application development. With only 10 billion activated parameters, it delivers a major jump in real-world...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 196608,
"max_completion_tokens": 196608,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-23",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.3,
"output": 1.2
},
"limit": {
"context": 204800,
"output": 131072
}
}
},
{
"id": "minimax/minimax-m2.5",
"name": "MiniMax M2.5",
"provider": "openrouter",
"family": "minimax",
"created_at": "2026-02-12 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2,
"cache_read_input_per_million": 0.03
}
}
},
"metadata": {
"description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 196608,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"parallel_tool_calls",
"presence_penalty",
"reasoning",
"reasoning_effort",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-12",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.3,
"output": 1.2,
"cache_read": 0.03
},
"limit": {
"context": 204800,
"output": 131072
}
}
},
{
"id": "minimax/minimax-m2.5:free",
"name": "MiniMax M2.5 (free)",
"provider": "openrouter",
"family": "minimax",
"created_at": "2026-02-12 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "MiniMax-M2.5 is a SOTA large language model designed for real-world productivity. Trained in a diverse range of complex real-world digital working environments, M2.5 builds upon the coding expertise of M2.1...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 196608,
"max_completion_tokens": 8192,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"temperature",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-12",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 204800,
"output": 131072
}
}
},
{
"id": "minimax/minimax-m2.7",
"name": "MiniMax M2.7",
"provider": "openrouter",
"family": "minimax",
"created_at": "2026-03-18 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2,
"cache_read_input_per_million": 0.06,
"cache_write_input_per_million": 0.375
}
}
},
"metadata": {
"description": "MiniMax-M2.7 is a next-generation large language model designed for autonomous, real-world productivity and continuous improvement. Built to actively participate in its own evolution, M2.7 integrates advanced agentic capabilities through multi-agent...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 196608,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-18",
"cost": {
"input": 0.3,
"output": 1.2,
"cache_read": 0.06,
"cache_write": 0.375
},
"limit": {
"context": 204800,
"output": 131072
}
}
},
{
"id": "mistralai/codestral-2508",
"name": "Codestral 2508",
"provider": "openrouter",
"family": "codestral",
"created_at": "2025-08-01 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.9
}
}
},
"metadata": {
"description": "Mistral's cutting-edge language model for coding released end of July 2025. Codestral specializes in low-latency, high-frequency tasks such as fill-in-the-middle (FIM), code correction and test generation.\n\n[Blog Post](https://mistral.ai/news/codestral-25-08)",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-01",
"cost": {
"input": 0.3,
"output": 0.9
},
"limit": {
"context": 256000,
"output": 256000
},
"knowledge": "2025-05"
}
},
{
"id": "mistralai/devstral-2512",
"name": "Devstral 2 2512",
"provider": "openrouter",
"family": "devstral",
"created_at": "2025-09-12 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-12",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-12"
}
},
{
"id": "mistralai/devstral-medium",
"name": "Mistral: Devstral Medium",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-07-10 15:28:41 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.39999999999999997,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.04
}
}
},
"metadata": {
"description": "Devstral Medium is a high-performance code generation and agentic reasoning model developed jointly by Mistral AI and All Hands AI. Positioned as a step up from Devstral Small, it achieves...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/devstral-medium-2507",
"name": "Devstral Medium",
"provider": "openrouter",
"family": "devstral",
"created_at": "2025-07-10 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-10",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2025-05"
}
},
{
"id": "mistralai/devstral-small",
"name": "Mistral: Devstral Small 1.1",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-07-10 15:19:11 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.3,
"cache_read_input_per_million": 0.01
}
}
},
"metadata": {
"description": "Devstral Small 1.1 is a 24B parameter open-weight language model for software engineering agents, developed by Mistral AI in collaboration with All Hands AI. Finetuned from Mistral Small 3.1 and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/devstral-small-2505",
"name": "Devstral Small",
"provider": "openrouter",
"family": "devstral",
"created_at": "2025-05-07 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.12
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-05-07",
"cost": {
"input": 0.06,
"output": 0.12
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2025-05"
}
},
{
"id": "mistralai/devstral-small-2507",
"name": "Devstral Small 1.1",
"provider": "openrouter",
"family": "devstral",
"created_at": "2025-07-10 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-10",
"cost": {
"input": 0.1,
"output": 0.3
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2025-05"
}
},
{
"id": "mistralai/ministral-14b-2512",
"name": "Mistral: Ministral 3 14B 2512",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-12-02 13:22:15 UTC",
"context_window": 262144,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.19999999999999998,
"output_per_million": 0.19999999999999998,
"cache_read_input_per_million": 0.02
}
}
},
"metadata": {
"description": "The largest model in the Ministral 3 family, Ministral 3 14B offers frontier capabilities and performance comparable to its larger Mistral Small 3.2 24B counterpart. A powerful and efficient language...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logprobs",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "mistralai/ministral-3b-2512",
"name": "Mistral: Ministral 3 3B 2512",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-12-02 13:19:20 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.09999999999999999,
"cache_read_input_per_million": 0.01
}
}
},
"metadata": {
"description": "The smallest model in the Ministral 3 family, Ministral 3 3B is a powerful, efficient tiny language model with vision capabilities.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logprobs",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "mistralai/ministral-8b-2512",
"name": "Mistral: Ministral 3 8B 2512",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-12-02 13:20:54 UTC",
"context_window": 262144,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.15,
"cache_read_input_per_million": 0.015
}
}
},
"metadata": {
"description": "A balanced model in the Ministral 3 family, Ministral 3 8B is a powerful, efficient tiny language model with vision capabilities.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logprobs",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "mistralai/mistral-7b-instruct-v0.1",
"name": "Mistral: Mistral 7B Instruct v0.1",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2023-09-28 00:00:00 UTC",
"context_window": 2824,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.11,
"output_per_million": 0.19
}
}
},
"metadata": {
"description": "A 7.3B parameter model that outperforms Llama 2 13B on all benchmarks, with optimizations for speed and context length.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": "mistral"
},
"top_provider": {
"context_length": 2824,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "mistralai/mistral-large",
"name": "Mistral Large",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2024-02-26 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 6.0,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "This is Mistral AI's flagship model, Mistral Large 2 (version `mistral-large-2407`). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mistral-large-2407",
"name": "Mistral Large 2407",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2024-11-19 01:06:55 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 6.0,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "This is Mistral AI's flagship model, Mistral Large 2 (version mistral-large-2407). It's a proprietary weights-available model and excels at reasoning, code, JSON, chat, and more. Read the launch announcement [here](https://mistral.ai/news/mistral-large-2407/)....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mistral-large-2411",
"name": "Mistral Large 2411",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2024-11-19 01:11:25 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 6.0,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Mistral Large 2 2411 is an update of [Mistral Large 2](/mistralai/mistral-large) released together with [Pixtral Large 2411](/mistralai/pixtral-large-2411) It provides a significant upgrade on the previous [Mistral Large 24.07](/mistralai/mistral-large-2407), with notable...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mistral-large-2512",
"name": "Mistral: Mistral Large 3 2512",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-12-01 21:27:52 UTC",
"context_window": 262144,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5,
"cache_read_input_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "Mistral Large 3 2512 is Mistral’s most capable model to date, featuring a sparse mixture-of-experts architecture with 41B active parameters (675B total), and released under the Apache 2.0 license.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mistral-medium-3",
"name": "Mistral Medium 3",
"provider": "openrouter",
"family": "mistral-medium",
"created_at": "2025-05-07 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"description": "Mistral Medium 3 is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances state-of-the-art reasoning and multimodal performance with 8× lower cost...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-07",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2025-05"
}
},
{
"id": "mistralai/mistral-medium-3-5",
"name": "Mistral: Mistral Medium 3.5",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2026-04-30 17:33:59 UTC",
"context_window": 262144,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.5,
"output_per_million": 7.5
}
}
},
"metadata": {
"description": "Mistral Medium 3.5 is a dense 128B instruction-following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mistral-medium-3.1",
"name": "Mistral Medium 3.1",
"provider": "openrouter",
"family": "mistral-medium",
"created_at": "2025-08-12 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2
}
}
},
"metadata": {
"description": "Mistral Medium 3.1 is an updated version of Mistral Medium 3, which is a high-performance enterprise-grade language model designed to deliver frontier-level capabilities at significantly reduced operational cost. It balances...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-12",
"cost": {
"input": 0.4,
"output": 2
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-05"
}
},
{
"id": "mistralai/mistral-nemo",
"name": "Mistral: Mistral Nemo",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2024-07-19 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.02,
"output_per_million": 0.03
}
}
},
"metadata": {
"description": "A 12B parameter model with a 128k token context length built by Mistral in collaboration with NVIDIA. The model is multilingual, supporting English, French, German, Spanish, Italian, Portuguese, Chinese, Japanese,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": "mistral"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "mistralai/mistral-saba",
"name": "Mistral: Saba",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-02-17 14:40:39 UTC",
"context_window": 32768,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.19999999999999998,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.02
}
}
},
"metadata": {
"description": "Mistral Saba is a 24B-parameter language model specifically designed for the Middle East and South Asia, delivering accurate and contextually relevant responses while maintaining efficient performance. Trained on curated regional...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mistral-small-24b-instruct-2501",
"name": "Mistral: Mistral Small 3",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-01-30 16:43:29 UTC",
"context_window": 32768,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.049999999999999996,
"output_per_million": 0.08
}
}
},
"metadata": {
"description": "Mistral Small 3 is a 24B-parameter language model optimized for low-latency performance across common AI tasks. Released under the Apache 2.0 license, it features both pre-trained and instruction-tuned versions designed...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "mistralai/mistral-small-2603",
"name": "Mistral Small 4",
"provider": "openrouter",
"family": "mistral-small",
"created_at": "2026-03-16 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "Mistral Small 4 is the next major release in the Mistral Small family, unifying the capabilities of several flagship Mistral models into a single system. It combines strong reasoning from...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-16",
"cost": {
"input": 0.15,
"output": 0.6
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-06"
}
},
{
"id": "mistralai/mistral-small-3.1-24b-instruct",
"name": "Mistral Small 3.1 24B Instruct",
"provider": "openrouter",
"family": "mistral-small",
"created_at": "2025-03-17 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.35,
"output_per_million": 0.56
}
}
},
"metadata": {
"description": "Mistral Small 3.1 24B Instruct is an upgraded variant of Mistral Small 3 (2501), featuring 24 billion parameters with advanced multimodal capabilities. It provides state-of-the-art performance in text-based reasoning and...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-03-17",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 128000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "mistralai/mistral-small-3.2-24b-instruct",
"name": "Mistral Small 3.2 24B Instruct",
"provider": "openrouter",
"family": "mistral-small",
"created_at": "2025-06-20 00:00:00 UTC",
"context_window": 96000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Mistral-Small-3.2-24B-Instruct-2506 is an updated 24B parameter model from Mistral optimized for instruction following, repetition reduction, and improved function calling. Compared to the 3.1 release, version 3.2 significantly improves accuracy on...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-20",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 96000,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "mistralai/mixtral-8x22b-instruct",
"name": "Mistral: Mixtral 8x22B Instruct",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2024-04-17 00:00:00 UTC",
"context_window": 65536,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 6.0,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Mistral's official instruct fine-tuned version of [Mixtral 8x22B](/models/mistralai/mixtral-8x22b). It uses 39B active parameters out of 141B, offering unparalleled cost efficiency for its size. Its strengths include: - strong math, coding,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": "mistral"
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/mixtral-8x7b-instruct",
"name": "Mistral: Mixtral 8x7B Instruct",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2023-12-10 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.54,
"output_per_million": 0.54
}
}
},
"metadata": {
"description": "Mixtral 8x7B Instruct is a pretrained generative Sparse Mixture of Experts, by Mistral AI, for chat and instruction use. Incorporates 8 experts (feed-forward networks) for a total of 47 billion...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": "mistral"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "mistralai/pixtral-large-2411",
"name": "Mistral: Pixtral Large 2411",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2024-11-19 00:49:48 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 6.0,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Pixtral Large is a 124B parameter, open-weight, multimodal model built on top of [Mistral Large 2](/mistralai/mistral-large-2411). The model is able to understand documents, charts and natural images. The model is...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "mistralai/voxtral-small-24b-2507",
"name": "Mistral: Voxtral Small 24B 2507",
"provider": "openrouter",
"family": "mistralai",
"created_at": "2025-10-30 14:39:04 UTC",
"context_window": 32000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"audio"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.3,
"cache_read_input_per_million": 0.01
}
}
},
"metadata": {
"description": "Voxtral Small is an enhancement of Mistral Small 3, incorporating state-of-the-art audio input capabilities while retaining best-in-class text performance. It excels at speech transcription, translation and audio understanding. Input audio...",
"architecture": {
"modality": "text+audio->text",
"input_modalities": [
"text",
"audio"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": null
},
"top_provider": {
"context_length": 32000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "moonshotai/kimi-k2",
"name": "Kimi K2",
"provider": "openrouter",
"family": "kimi",
"created_at": "2025-07-11 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.55,
"output_per_million": 2.2
}
}
},
"metadata": {
"description": "Kimi K2 Instruct is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32 billion active per forward pass. It is optimized for...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-11",
"cost": {
"input": 0.55,
"output": 2.2
},
"limit": {
"context": 131072,
"output": 32768
},
"knowledge": "2024-10"
}
},
{
"id": "moonshotai/kimi-k2-0905",
"name": "Kimi K2 Instruct 0905",
"provider": "openrouter",
"family": "kimi",
"created_at": "2025-09-05 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.5
}
}
},
"metadata": {
"description": "Kimi K2 0905 is the September update of [Kimi K2 0711](moonshotai/kimi-k2). It is a large-scale Mixture-of-Experts (MoE) language model developed by Moonshot AI, featuring 1 trillion total parameters with 32...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-05",
"cost": {
"input": 0.6,
"output": 2.5
},
"limit": {
"context": 262144,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "moonshotai/kimi-k2-0905:exacto",
"name": "Kimi K2 Instruct 0905 (exacto)",
"provider": "openrouter",
"family": "kimi",
"created_at": "2025-09-05 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.5
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-05",
"cost": {
"input": 0.6,
"output": 2.5
},
"limit": {
"context": 262144,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "moonshotai/kimi-k2-thinking",
"name": "Kimi K2 Thinking",
"provider": "openrouter",
"family": "kimi-thinking",
"created_at": "2025-11-06 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.15
}
}
},
"metadata": {
"description": "Kimi K2 Thinking is Moonshot AI’s most advanced open reasoning model to date, extending the K2 series into agentic, long-horizon reasoning. Built on the trillion-parameter Mixture-of-Experts (MoE) architecture introduced in...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-11-06",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.6,
"output": 2.5,
"cache_read": 0.15
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2024-08"
}
},
{
"id": "moonshotai/kimi-k2.5",
"name": "Kimi K2.5",
"provider": "openrouter",
"family": "kimi",
"created_at": "2026-01-27 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 3,
"cache_read_input_per_million": 0.1
}
}
},
"metadata": {
"description": "Kimi K2.5 is Moonshot AI's native multimodal model, delivering state-of-the-art visual coding capability and a self-directed agent swarm paradigm. Built on Kimi K2 with continued pretraining over approximately 15T mixed...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65535,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"parallel_tool_calls",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-01-27",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.6,
"output": 3,
"cache_read": 0.1
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-01"
}
},
{
"id": "moonshotai/kimi-k2.6",
"name": "Kimi K2.6",
"provider": "openrouter",
"family": "kimi",
"created_at": "2026-04-20 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.95,
"output_per_million": 4,
"cache_read_input_per_million": 0.16
}
}
},
"metadata": {
"description": "Kimi K2.6 is Moonshot AI's next-generation multimodal model, designed for long-horizon coding, coding-driven UI/UX generation, and multi-agent orchestration. It handles complex end-to-end coding tasks across Python, Rust, and Go, and...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"parallel_tool_calls",
"presence_penalty",
"reasoning",
"reasoning_effort",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-20",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.95,
"output": 4,
"cache_read": 0.16
},
"limit": {
"context": 262144,
"output": 262144
}
}
},
{
"id": "morph/morph-v3-fast",
"name": "Morph: Morph V3 Fast",
"provider": "openrouter",
"family": "morph",
"created_at": "2025-07-07 17:40:02 UTC",
"context_window": 81920,
"max_output_tokens": 38000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.7999999999999999,
"output_per_million": 1.2
}
}
},
"metadata": {
"description": "Morph's fastest apply model for code edits. ~10,500 tokens/sec with 96% accuracy for rapid code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code} {edit_snippet}...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 81920,
"max_completion_tokens": 38000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature"
]
}
},
{
"id": "morph/morph-v3-large",
"name": "Morph: Morph V3 Large",
"provider": "openrouter",
"family": "morph",
"created_at": "2025-07-07 17:54:18 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8999999999999999,
"output_per_million": 1.9
}
}
},
"metadata": {
"description": "Morph's high-accuracy apply model for complex code edits. ~4,500 tokens/sec with 98% accuracy for precise code transformations. The model requires the prompt to be in the following format: {instruction} {initial_code}...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature"
]
}
},
{
"id": "nex-agi/deepseek-v3.1-nex-n1",
"name": "Nex AGI: DeepSeek V3.1 Nex N1",
"provider": "openrouter",
"family": "nex-agi",
"created_at": "2025-12-08 14:33:13 UTC",
"context_window": 131072,
"max_output_tokens": 163840,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.135,
"output_per_million": 0.5
}
}
},
"metadata": {
"description": "DeepSeek V3.1 Nex-N1 is the flagship release of the Nex-N1 series — a post-trained model designed to highlight agent autonomy, tool use, and real-world productivity. Nex-N1 demonstrates competitive performance across...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 163840,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"response_format",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "nousresearch/hermes-2-pro-llama-3-8b",
"name": "NousResearch: Hermes 2 Pro - Llama-3 8B",
"provider": "openrouter",
"family": "nousresearch",
"created_at": "2024-05-27 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.14
}
}
},
"metadata": {
"description": "Hermes 2 Pro is an upgraded, retrained version of Nous Hermes 2, consisting of an updated and cleaned version of the OpenHermes 2.5 Dataset, as well as a newly introduced...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 8192,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "nousresearch/hermes-3-llama-3.1-405b",
"name": "Nous: Hermes 3 405B Instruct",
"provider": "openrouter",
"family": "nousresearch",
"created_at": "2024-08-16 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.0,
"output_per_million": 1.0
}
}
},
"metadata": {
"description": "Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "nousresearch/hermes-3-llama-3.1-405b:free",
"name": "Hermes 3 405B Instruct (free)",
"provider": "openrouter",
"family": "hermes",
"created_at": "2024-08-16 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Hermes 3 is a generalist language model with many improvements over Hermes 2, including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"stop",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-08-16",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2023-12"
}
},
{
"id": "nousresearch/hermes-3-llama-3.1-70b",
"name": "Nous: Hermes 3 70B Instruct",
"provider": "openrouter",
"family": "nousresearch",
"created_at": "2024-08-18 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.3
}
}
},
"metadata": {
"description": "Hermes 3 is a generalist language model with many improvements over [Hermes 2](/models/nousresearch/nous-hermes-2-mistral-7b-dpo), including advanced agentic capabilities, much better roleplaying, reasoning, multi-turn conversation, long context coherence, and improvements across the...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "nousresearch/hermes-4-405b",
"name": "Hermes 4 405B",
"provider": "openrouter",
"family": "hermes",
"created_at": "2025-08-25 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 3
}
}
},
"metadata": {
"description": "Hermes 4 is a large-scale reasoning model built on Meta-Llama-3.1-405B and released by Nous Research. It introduces a hybrid reasoning mode, where the model can choose to deliberate internally with...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-25",
"cost": {
"input": 1,
"output": 3
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2023-12"
}
},
{
"id": "nousresearch/hermes-4-70b",
"name": "Hermes 4 70B",
"provider": "openrouter",
"family": "hermes",
"created_at": "2025-08-25 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.13,
"output_per_million": 0.4
}
}
},
"metadata": {
"description": "Hermes 4 70B is a hybrid reasoning model from Nous Research, built on Meta-Llama-3.1-70B. It introduces the same hybrid mode as the larger 405B release, allowing the model to either...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-25",
"cost": {
"input": 0.13,
"output": 0.4
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2023-12"
}
},
{
"id": "nvidia/llama-3.1-nemotron-70b-instruct",
"name": "NVIDIA: Llama 3.1 Nemotron 70B Instruct",
"provider": "openrouter",
"family": "nvidia",
"created_at": "2024-10-15 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.2,
"output_per_million": 1.2
}
}
},
"metadata": {
"description": "NVIDIA's Llama 3.1 Nemotron 70B is a language model designed for generating precise and useful responses. Leveraging [Llama 3.1 70B](/models/meta-llama/llama-3.1-70b-instruct) architecture and Reinforcement Learning from Human Feedback (RLHF), it excels...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "nvidia/llama-3.3-nemotron-super-49b-v1.5",
"name": "NVIDIA: Llama 3.3 Nemotron Super 49B V1.5",
"provider": "openrouter",
"family": "nvidia",
"created_at": "2025-10-10 13:03:15 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.39999999999999997
}
}
},
"metadata": {
"description": "Llama-3.3-Nemotron-Super-49B-v1.5 is a 49B-parameter, English-centric reasoning/chat model derived from Meta’s Llama-3.3-70B-Instruct with a 128K context. It’s post-trained for agentic workflows (RAG, tool calling) via SFT across math, code, science, and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "nvidia/nemotron-3-nano-30b-a3b",
"name": "NVIDIA: Nemotron 3 Nano 30B A3B",
"provider": "openrouter",
"family": "nvidia",
"created_at": "2025-12-14 16:54:35 UTC",
"context_window": 262144,
"max_output_tokens": 228000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.049999999999999996,
"output_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 228000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "nvidia/nemotron-3-nano-30b-a3b:free",
"name": "Nemotron 3 Nano 30B A3B (free)",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2025-12-14 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 256000,
"output": 256000
},
"knowledge": "2025-11"
}
},
{
"id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free",
"name": "Nemotron 3 Nano Omni (free)",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2026-04-28 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {},
"metadata": {
"description": "NVIDIA Nemotron™ 3 Nano Omni is a 30B-A3B open multimodal model designed to function as a perception and context sub-agent in enterprise agent systems. It accepts text, image, video, and...",
"architecture": {
"modality": "text+image+audio+video->text",
"input_modalities": [
"text",
"audio",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 256000,
"output": 65536
}
}
},
{
"id": "nvidia/nemotron-3-super-120b-a12b",
"name": "Nemotron 3 Super",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2026-03-11 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.5
}
}
},
"metadata": {
"description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-11",
"cost": {
"input": 0.1,
"output": 0.5
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2024-04"
}
},
{
"id": "nvidia/nemotron-3-super-120b-a12b:free",
"name": "Nemotron 3 Super (free)",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2026-03-11 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {},
"metadata": {
"description": "NVIDIA Nemotron 3 Super is a 120B-parameter open hybrid MoE model, activating just 12B parameters for maximum compute efficiency and accuracy in complex multi-agent applications. Built on a hybrid Mamba-Transformer...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-11",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2024-04"
}
},
{
"id": "nvidia/nemotron-nano-12b-v2-vl",
"name": "NVIDIA: Nemotron Nano 12B 2 VL",
"provider": "openrouter",
"family": "nvidia",
"created_at": "2025-10-28 18:19:25 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.19999999999999998,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "nvidia/nemotron-nano-12b-v2-vl:free",
"name": "Nemotron Nano 12B 2 VL (free)",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2025-10-28 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {},
"metadata": {
"description": "NVIDIA Nemotron Nano 2 VL is a 12-billion-parameter open multimodal reasoning model designed for video understanding and document intelligence. It introduces a hybrid Transformer-Mamba architecture, combining transformer-level accuracy with Mamba’s...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2025-11"
}
},
{
"id": "nvidia/nemotron-nano-9b-v2",
"name": "nvidia-nemotron-nano-9b-v2",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2025-08-18 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.16
}
}
},
"metadata": {
"description": "NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-18",
"cost": {
"input": 0.04,
"output": 0.16
},
"limit": {
"context": 131072,
"output": 131072
},
"knowledge": "2024-09"
}
},
{
"id": "nvidia/nemotron-nano-9b-v2:free",
"name": "Nemotron Nano 9B V2 (free)",
"provider": "openrouter",
"family": "nemotron",
"created_at": "2025-09-05 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "NVIDIA-Nemotron-Nano-9B-v2 is a large language model (LLM) trained from scratch by NVIDIA, and designed as a unified model for both reasoning and non-reasoning tasks. It responds to user queries and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-18",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 128000,
"output": 128000
},
"knowledge": "2024-09"
}
},
{
"id": "openai/gpt-3.5-turbo",
"name": "OpenAI: GPT-3.5 Turbo",
"provider": "openrouter",
"family": "openai",
"created_at": "2023-05-28 00:00:00 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 1.5
}
}
},
"metadata": {
"description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 16385,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-3.5-turbo-0613",
"name": "OpenAI: GPT-3.5 Turbo (older v0613)",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-01-25 00:00:00 UTC",
"context_window": 4095,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.0,
"output_per_million": 2.0
}
}
},
"metadata": {
"description": "GPT-3.5 Turbo is OpenAI's fastest model. It can understand and generate natural language or code, and is optimized for chat and traditional completion tasks.\n\nTraining data up to Sep 2021.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 4095,
"max_completion_tokens": 4096,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-3.5-turbo-16k",
"name": "OpenAI: GPT-3.5 Turbo 16k",
"provider": "openrouter",
"family": "openai",
"created_at": "2023-08-28 00:00:00 UTC",
"context_window": 16385,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 4.0
}
}
},
"metadata": {
"description": "This model offers four times the context length of gpt-3.5-turbo, allowing it to support approximately 20 pages of text in a single request at a higher cost. Training data: up...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 16385,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-3.5-turbo-instruct",
"name": "OpenAI: GPT-3.5 Turbo Instruct",
"provider": "openrouter",
"family": "openai",
"created_at": "2023-09-28 00:00:00 UTC",
"context_window": 4095,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.5,
"output_per_million": 2.0
}
}
},
"metadata": {
"description": "This model is a variant of GPT-3.5 Turbo tuned for instructional prompts and omitting chat-related optimizations. Training data: up to Sep 2021.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 4095,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4",
"name": "OpenAI: GPT-4",
"provider": "openrouter",
"family": "openai",
"created_at": "2023-05-28 00:00:00 UTC",
"context_window": 8191,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30.0,
"output_per_million": 60.0
}
}
},
"metadata": {
"description": "OpenAI's flagship model, GPT-4 is a large-scale multimodal language model capable of solving difficult problems with greater accuracy than previous models due to its broader general knowledge and advanced reasoning...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 8191,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4-0314",
"name": "OpenAI: GPT-4 (older v0314)",
"provider": "openrouter",
"family": "openai",
"created_at": "2023-05-28 00:00:00 UTC",
"context_window": 8191,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30.0,
"output_per_million": 60.0
}
}
},
"metadata": {
"description": "GPT-4-0314 is the first version of GPT-4 released, with a context length of 8,192 tokens, and was supported until June 14. Training data: up to Sep 2021.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 8191,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4-1106-preview",
"name": "OpenAI: GPT-4 Turbo (older v1106)",
"provider": "openrouter",
"family": "openai",
"created_at": "2023-11-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to April 2023.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4-turbo",
"name": "OpenAI: GPT-4 Turbo",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-04-09 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"description": "The latest GPT-4 Turbo model with vision capabilities. Vision requests can now use JSON mode and function calling.\n\nTraining data: up to December 2023.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4-turbo-preview",
"name": "OpenAI: GPT-4 Turbo Preview",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-01-25 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 30.0
}
}
},
"metadata": {
"description": "The preview GPT-4 model with improved instruction following, JSON mode, reproducible outputs, parallel function calling, and more. Training data: up to Dec 2023. **Note:** heavily rate limited by OpenAI while...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4096,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4.1",
"name": "GPT-4.1",
"provider": "openrouter",
"family": "gpt",
"created_at": "2025-04-14 00:00:00 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 8,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "GPT-4.1 is a flagship large language model optimized for advanced instruction following, real-world software engineering, and long-context reasoning. It supports a 1 million token context window and outperforms GPT-4o and...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1047576,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_completion_tokens",
"max_tokens",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-14",
"cost": {
"input": 2,
"output": 8,
"cache_read": 0.5
},
"limit": {
"context": 1047576,
"output": 32768
},
"knowledge": "2024-04"
}
},
{
"id": "openai/gpt-4.1-mini",
"name": "GPT-4.1 Mini",
"provider": "openrouter",
"family": "gpt-mini",
"created_at": "2025-04-14 00:00:00 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 1.6,
"cache_read_input_per_million": 0.1
}
}
},
"metadata": {
"description": "GPT-4.1 Mini is a mid-sized model delivering performance competitive with GPT-4o at substantially lower latency and cost. It retains a 1 million token context window and scores 45.1% on hard...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1047576,
"max_completion_tokens": 32768,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_completion_tokens",
"max_tokens",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-14",
"cost": {
"input": 0.4,
"output": 1.6,
"cache_read": 0.1
},
"limit": {
"context": 1047576,
"output": 32768
},
"knowledge": "2024-04"
}
},
{
"id": "openai/gpt-4.1-nano",
"name": "OpenAI: GPT-4.1 Nano",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-04-14 17:22:49 UTC",
"context_window": 1047576,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.39999999999999997,
"cache_read_input_per_million": 0.024999999999999998
}
}
},
"metadata": {
"description": "For tasks that demand low latency, GPT‑4.1 nano is the fastest and cheapest model in the GPT-4.1 series. It delivers exceptional performance at a small size with its 1 million...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1047576,
"max_completion_tokens": 32768,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_completion_tokens",
"max_tokens",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "openai/gpt-4o",
"name": "OpenAI: GPT-4o",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-05-13 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p",
"web_search_options"
]
}
},
{
"id": "openai/gpt-4o-2024-05-13",
"name": "OpenAI: GPT-4o (2024-05-13)",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-05-13 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 15.0
}
}
},
"metadata": {
"description": "GPT-4o (\"o\" for \"omni\") is OpenAI's latest AI model, supporting both text and image inputs with text outputs. It maintains the intelligence level of [GPT-4 Turbo](/models/openai/gpt-4-turbo) while being twice as...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 4096,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p",
"web_search_options"
]
}
},
{
"id": "openai/gpt-4o-2024-08-06",
"name": "OpenAI: GPT-4o (2024-08-06)",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-08-06 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"description": "The 2024-08-06 version of GPT-4o offers improved performance in structured outputs, with the ability to supply a JSON schema in the respone_format. Read more [here](https://openai.com/index/introducing-structured-outputs-in-the-api/). GPT-4o (\"o\" for \"omni\") is...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p",
"web_search_options"
]
}
},
{
"id": "openai/gpt-4o-2024-11-20",
"name": "OpenAI: GPT-4o (2024-11-20)",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-11-20 18:33:14 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"description": "The 2024-11-20 version of GPT-4o offers a leveled-up creative writing ability with more natural, engaging, and tailored writing to improve relevance & readability. It’s also better at working with uploaded...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p",
"web_search_options"
]
}
},
{
"id": "openai/gpt-4o-audio-preview",
"name": "OpenAI: GPT-4o Audio",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-08-15 04:44:21 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"audio",
"text"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "The gpt-4o-audio-preview model adds support for audio inputs as prompts. This enhancement allows the model to detect nuances within audio recordings and add depth to generated user experiences. Audio outputs...",
"architecture": {
"modality": "text+audio->text+audio",
"input_modalities": [
"audio",
"text"
],
"output_modalities": [
"text",
"audio"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-4o-mini",
"name": "GPT-4o-mini",
"provider": "openrouter",
"family": "gpt-mini",
"created_at": "2024-07-18 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.08
}
}
},
"metadata": {
"description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p",
"web_search_options"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-07-18",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.08
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2024-10"
}
},
{
"id": "openai/gpt-4o-mini-2024-07-18",
"name": "OpenAI: GPT-4o-mini (2024-07-18)",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-07-18 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.075
}
}
},
"metadata": {
"description": "GPT-4o mini is OpenAI's newest model after [GPT-4 Omni](/models/openai/gpt-4o), supporting both text and image inputs with text outputs. As their most advanced small model, it is many multiples more affordable...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p",
"web_search_options"
]
}
},
{
"id": "openai/gpt-4o-mini-search-preview",
"name": "OpenAI: GPT-4o-mini Search Preview",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-03-12 22:22:02 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6
}
}
},
"metadata": {
"description": "GPT-4o mini Search Preview is a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"structured_outputs",
"web_search_options"
]
}
},
{
"id": "openai/gpt-4o-search-preview",
"name": "OpenAI: GPT-4o Search Preview",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-03-12 22:19:09 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "GPT-4o Search Previewis a specialized model for web search in Chat Completions. It is trained to understand and execute web search queries.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"structured_outputs",
"web_search_options"
]
}
},
{
"id": "openai/gpt-5",
"name": "GPT-5",
"provider": "openrouter",
"family": "gpt",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-10-01",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10
}
}
},
"metadata": {
"description": "GPT-5 is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and accuracy...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-07",
"cost": {
"input": 1.25,
"output": 10
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-10-01"
}
},
{
"id": "openai/gpt-5-chat",
"name": "GPT-5 Chat (latest)",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10
}
}
},
"metadata": {
"description": "GPT-5 Chat is designed for advanced, natural, multimodal, and context-aware conversations for enterprise applications.",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"response_format",
"seed",
"structured_outputs"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-07",
"cost": {
"input": 1.25,
"output": 10
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5-codex",
"name": "GPT-5 Codex",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-09-15 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-10-01",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"description": "GPT-5-Codex is a specialized version of GPT-5 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-15",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-10-01"
}
},
{
"id": "openai/gpt-5-image",
"name": "GPT-5 Image",
"provider": "openrouter",
"family": "gpt",
"created_at": "2025-10-14 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-10-01",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 10,
"cache_read_input_per_million": 1.25
}
}
},
"metadata": {
"description": "[GPT-5](https://openrouter.ai/openai/gpt-5) Image combines OpenAI's GPT-5 model with state-of-the-art image generation capabilities. It offers major improvements in reasoning, code quality, and user experience while incorporating GPT Image 1's superior instruction following,...",
"architecture": {
"modality": "text+image+file->text+image",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"image",
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-14",
"cost": {
"input": 5,
"output": 10,
"cache_read": 1.25
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-10-01"
}
},
{
"id": "openai/gpt-5-image-mini",
"name": "OpenAI: GPT-5 Image Mini",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-10-16 14:23:03 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"file",
"image",
"text"
],
"output": [
"image",
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 2.0,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"description": "GPT-5 Image Mini combines OpenAI's advanced language capabilities, powered by [GPT-5 Mini](https://openrouter.ai/openai/gpt-5-mini), with GPT Image 1 Mini for efficient image generation. This natively multimodal model features superior instruction following, text...",
"architecture": {
"modality": "text+image+file->text+image",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"image",
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-5-mini",
"name": "GPT-5 Mini",
"provider": "openrouter",
"family": "gpt-mini",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-10-01",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2
}
}
},
"metadata": {
"description": "GPT-5 Mini is a compact version of GPT-5, designed to handle lighter-weight reasoning tasks. It provides the same instruction-following and safety-tuning benefits as GPT-5, but with reduced latency and cost....",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-07",
"cost": {
"input": 0.25,
"output": 2
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-10-01"
}
},
{
"id": "openai/gpt-5-nano",
"name": "GPT-5 Nano",
"provider": "openrouter",
"family": "gpt-nano",
"created_at": "2025-08-07 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-10-01",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.4
}
}
},
"metadata": {
"description": "GPT-5-Nano is the smallest and fastest variant in the GPT-5 system, optimized for developer tools, rapid interactions, and ultra-low latency environments. While limited in reasoning depth compared to its larger...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-07",
"cost": {
"input": 0.05,
"output": 0.4
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-10-01"
}
},
{
"id": "openai/gpt-5-pro",
"name": "GPT-5 Pro",
"provider": "openrouter",
"family": "gpt-pro",
"created_at": "2025-10-06 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 272000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 120
}
}
},
"metadata": {
"description": "GPT-5 Pro is OpenAI’s most advanced model, offering major improvements in reasoning, code quality, and user experience. It is optimized for complex tasks that require step-by-step reasoning, instruction following, and...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-10-06",
"cost": {
"input": 15,
"output": 120
},
"limit": {
"context": 400000,
"output": 272000
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5.1",
"name": "GPT-5.1",
"provider": "openrouter",
"family": "gpt",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"description": "GPT-5.1 is the latest frontier-grade model in the GPT-5 series, offering stronger general-purpose reasoning, improved instruction adherence, and a more natural conversational style compared to GPT-5. It uses adaptive reasoning...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5.1-chat",
"name": "GPT-5.1 Chat",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"description": "GPT-5.1 Chat (AKA Instant is the fast, lightweight member of the 5.1 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_completion_tokens",
"max_tokens",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5.1-codex",
"name": "GPT-5.1-Codex",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"description": "GPT-5.1-Codex is a specialized version of GPT-5.1 optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-13",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5.1-codex-max",
"name": "GPT-5.1-Codex-Max",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 9,
"cache_read_input_per_million": 0.11
}
}
},
"metadata": {
"description": "GPT-5.1-Codex-Max is OpenAI’s latest agentic coding model, designed for long-running, high-context software development tasks. It is based on an updated version of the 5.1 reasoning stack and trained on agentic...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-13",
"cost": {
"input": 1.1,
"output": 9,
"cache_read": 0.11
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5.1-codex-mini",
"name": "GPT-5.1-Codex-Mini",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-11-13 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 100000,
"knowledge_cutoff": "2024-09-30",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 2,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"description": "GPT-5.1-Codex-Mini is a smaller and faster version of GPT-5.1-Codex",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-13",
"cost": {
"input": 0.25,
"output": 2,
"cache_read": 0.025
},
"limit": {
"context": 400000,
"output": 100000
},
"knowledge": "2024-09-30"
}
},
{
"id": "openai/gpt-5.2",
"name": "GPT-5.2",
"provider": "openrouter",
"family": "gpt",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"description": "GPT-5.2 is the latest frontier-grade model in the GPT-5 series, offering stronger agentic and long context perfomance compared to GPT-5.1. It uses adaptive reasoning to allocate computation dynamically, responding quickly...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.2-chat",
"name": "GPT-5.2 Chat",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"description": "GPT-5.2 Chat (AKA Instant) is the fast, lightweight member of the 5.2 family, optimized for low-latency chat while retaining strong general intelligence. It uses adaptive reasoning to selectively “think” on...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 32000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_completion_tokens",
"max_tokens",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 128000,
"output": 16384
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.2-codex",
"name": "GPT-5.2-Codex",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2026-01-14 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"description": "GPT-5.2-Codex is an upgraded version of GPT-5.1-Codex optimized for software engineering and coding workflows. It is designed for both interactive development sessions and long, independent execution of complex engineering tasks....",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-01-14",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.2-pro",
"name": "GPT-5.2 Pro",
"provider": "openrouter",
"family": "gpt-pro",
"created_at": "2025-12-11 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 21,
"output_per_million": 168
}
}
},
"metadata": {
"description": "GPT-5.2 Pro is OpenAI’s most advanced model, offering major improvements in agentic coding and long context performance over GPT-5 Pro. It is optimized for complex tasks that require step-by-step reasoning,...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2025-12-11",
"cost": {
"input": 21,
"output": 168
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.3-chat",
"name": "OpenAI: GPT-5.3 Chat",
"provider": "openrouter",
"family": "openai",
"created_at": "2026-03-03 18:54:21 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14.0,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"description": "GPT-5.3 Chat is an update to ChatGPT's most-used model that makes everyday conversations smoother, more useful, and more directly helpful. It delivers more accurate answers with better contextualization and significantly...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_completion_tokens",
"max_tokens",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openai/gpt-5.3-codex",
"name": "GPT-5.3-Codex",
"provider": "openrouter",
"family": "gpt-codex",
"created_at": "2026-02-24 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.75,
"output_per_million": 14,
"cache_read_input_per_million": 0.175
}
}
},
"metadata": {
"description": "GPT-5.3-Codex is OpenAI’s most advanced agentic coding model, combining the frontier software engineering performance of GPT-5.2-Codex with the broader reasoning and professional knowledge capabilities of GPT-5.2. It achieves state-of-the-art results...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-02-24",
"cost": {
"input": 1.75,
"output": 14,
"cache_read": 0.175
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.4",
"name": "GPT-5.4",
"provider": "openrouter",
"family": "gpt",
"created_at": "2026-03-05 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 15,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"description": "GPT-5.4 is OpenAI’s latest frontier model, unifying the Codex and GPT lines into a single system. It features a 1M+ token context window (922K input, 128K output) with support for...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1050000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-03-05",
"cost": {
"input": 2.5,
"output": 15,
"cache_read": 0.25,
"context_over_200k": {
"input": 5,
"output": 22.5,
"cache_read": 0.5
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.4-image-2",
"name": "OpenAI: GPT-5.4 Image 2",
"provider": "openrouter",
"family": "openai",
"created_at": "2026-04-21 18:52:08 UTC",
"context_window": 272000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"file"
],
"output": [
"image",
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 8.0,
"output_per_million": 15.0,
"cache_read_input_per_million": 2.0
}
}
},
"metadata": {
"description": "[GPT-5.4](https://openrouter.ai/openai/gpt-5.4) Image 2 combines OpenAI's GPT-5.4 model with state-of-the-art image generation capabilities from GPT Image 2. It enables rich multimodal workflows, allowing users to seamlessly move between reasoning, coding, and...",
"architecture": {
"modality": "text+image+file->text+image",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"image",
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 272000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"top_logprobs"
]
}
},
{
"id": "openai/gpt-5.4-mini",
"name": "GPT-5.4 Mini",
"provider": "openrouter",
"family": "gpt-mini",
"created_at": "2026-03-17 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.75,
"output_per_million": 4.5,
"cache_read_input_per_million": 0.075
}
}
},
"metadata": {
"description": "GPT-5.4 mini brings the core capabilities of GPT-5.4 to a faster, more efficient model optimized for high-throughput workloads. It supports text and image inputs with strong performance across reasoning, coding,...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-17",
"cost": {
"input": 0.75,
"output": 4.5,
"cache_read": 0.075
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.4-nano",
"name": "GPT-5.4 Nano",
"provider": "openrouter",
"family": "gpt-nano",
"created_at": "2026-03-17 00:00:00 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 1.25,
"cache_read_input_per_million": 0.02
}
}
},
"metadata": {
"description": "GPT-5.4 nano is the most lightweight and cost-efficient variant of the GPT-5.4 family, optimized for speed-critical and high-volume tasks. It supports text and image inputs and is designed for low-latency...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-17",
"cost": {
"input": 0.2,
"output": 1.25,
"cache_read": 0.02
},
"limit": {
"context": 400000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.4-pro",
"name": "GPT-5.4 Pro",
"provider": "openrouter",
"family": "gpt-pro",
"created_at": "2026-03-05 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30,
"output_per_million": 180,
"cache_read_input_per_million": 30
}
}
},
"metadata": {
"description": "GPT-5.4 Pro is OpenAI's most advanced model, building on GPT-5.4's unified architecture with enhanced reasoning capabilities for complex, high-stakes tasks. It features a 1M+ token context window (922K input, 128K...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1050000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-03-05",
"cost": {
"input": 30,
"output": 180,
"cache_read": 30
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-08-31"
}
},
{
"id": "openai/gpt-5.5",
"name": "GPT-5.5",
"provider": "openrouter",
"family": "gpt",
"created_at": "2026-04-23 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-12-01",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 30,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "GPT-5.5 is OpenAI’s frontier model designed for complex professional workloads, building on GPT-5.4 with stronger reasoning, higher reliability, and improved token efficiency on hard tasks. It features a 1M+ token...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1050000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-23",
"cost": {
"input": 5,
"output": 30,
"cache_read": 0.5,
"context_over_200k": {
"input": 10,
"output": 45,
"cache_read": 1
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-12-01"
}
},
{
"id": "openai/gpt-5.5-pro",
"name": "GPT-5.5 Pro",
"provider": "openrouter",
"family": "gpt-pro",
"created_at": "2026-04-23 00:00:00 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-12-01",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 30,
"output_per_million": 180
}
}
},
"metadata": {
"description": "GPT-5.5 Pro is OpenAI’s high-capability model optimized for deep reasoning and accuracy on complex, high-stakes workloads. It features a 1M+ token context window (922K input, 128K output) with support for...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 1050000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-23",
"cost": {
"input": 30,
"output": 180,
"context_over_200k": {
"input": 60,
"output": 270
}
},
"limit": {
"context": 1050000,
"input": 922000,
"output": 128000
},
"knowledge": "2025-12-01"
}
},
{
"id": "openai/gpt-audio",
"name": "OpenAI: GPT Audio",
"provider": "openrouter",
"family": "openai",
"created_at": "2026-01-19 22:42:49 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"audio"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.5,
"output_per_million": 10.0
}
}
},
"metadata": {
"description": "The gpt-audio model is OpenAI's first generally available audio model. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Audio is priced...",
"architecture": {
"modality": "text+audio->text+audio",
"input_modalities": [
"text",
"audio"
],
"output_modalities": [
"text",
"audio"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-audio-mini",
"name": "OpenAI: GPT Audio Mini",
"provider": "openrouter",
"family": "openai",
"created_at": "2026-01-19 21:50:19 UTC",
"context_window": 128000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"audio"
],
"output": [
"text",
"audio"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.4
}
}
},
"metadata": {
"description": "A cost-efficient version of GPT Audio. The new snapshot features an upgraded decoder for more natural sounding voices and maintains better voice consistency. Input is priced at $0.60 per million...",
"architecture": {
"modality": "text+audio->text+audio",
"input_modalities": [
"text",
"audio"
],
"output_modalities": [
"text",
"audio"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": 16384,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/gpt-chat-latest",
"name": "OpenAI: GPT Chat Latest",
"provider": "openrouter",
"family": "openai",
"created_at": "2026-05-05 16:56:52 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 30.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "GPT Chat Latest points to OpenAI's stable API alias `chat-latest` that always resolves to the latest Instant chat model used in ChatGPT. As OpenAI rolls out new Instant model updates...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"tool_choice",
"tools",
"top_logprobs"
]
}
},
{
"id": "openai/gpt-oss-120b",
"name": "GPT OSS 120B",
"provider": "openrouter",
"family": "gpt-oss",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.072,
"output_per_million": 0.28
}
}
},
"metadata": {
"description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 0.072,
"output": 0.28
},
"limit": {
"context": 131072,
"output": 32768
}
}
},
{
"id": "openai/gpt-oss-120b:exacto",
"name": "GPT OSS 120B (exacto)",
"provider": "openrouter",
"family": "gpt-oss",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.24
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 0.05,
"output": 0.24
},
"limit": {
"context": 131072,
"output": 32768
}
}
},
{
"id": "openai/gpt-oss-120b:free",
"name": "gpt-oss-120b (free)",
"provider": "openrouter",
"family": "gpt-oss",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "gpt-oss-120b is an open-weight, 117B-parameter Mixture-of-Experts (MoE) language model from OpenAI designed for high-reasoning, agentic, and general-purpose production use cases. It activates 5.1B parameters per forward pass and is optimized...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"seed",
"stop",
"temperature",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 32768
}
}
},
{
"id": "openai/gpt-oss-20b",
"name": "GPT OSS 20B",
"provider": "openrouter",
"family": "gpt-oss",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.05,
"output_per_million": 0.2
}
}
},
"metadata": {
"description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 0.05,
"output": 0.2
},
"limit": {
"context": 131072,
"output": 32768
}
}
},
{
"id": "openai/gpt-oss-20b:free",
"name": "gpt-oss-20b (free)",
"provider": "openrouter",
"family": "gpt-oss",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 8192,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"seed",
"stop",
"temperature",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-31",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 131072,
"output": 32768
}
}
},
{
"id": "openai/gpt-oss-safeguard-20b",
"name": "GPT OSS Safeguard 20B",
"provider": "openrouter",
"family": "gpt-oss",
"created_at": "2025-10-29 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"description": "gpt-oss-safeguard-20b is a safety reasoning model from OpenAI built upon gpt-oss-20b. This open-weight, 21B-parameter Mixture-of-Experts (MoE) model offers lower latency for safety tasks like content classification, LLM filtering, and trust...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-10-29",
"cost": {
"input": 0.075,
"output": 0.3
},
"limit": {
"context": 131072,
"output": 65536
}
}
},
{
"id": "openai/o1",
"name": "OpenAI: o1",
"provider": "openrouter",
"family": "openai",
"created_at": "2024-12-17 18:26:39 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15.0,
"output_per_million": 60.0,
"cache_read_input_per_million": 7.5
}
}
},
"metadata": {
"description": "The latest and strongest model family from OpenAI, o1 is designed to spend more time thinking before responding. The o1 model series is trained with large-scale reinforcement learning to reason...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openai/o1-pro",
"name": "OpenAI: o1-pro",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-03-19 22:26:51 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 150.0,
"output_per_million": 600.0
}
}
},
"metadata": {
"description": "The o1 series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o1-pro model uses more compute to think harder and provide...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs"
]
}
},
{
"id": "openai/o3",
"name": "OpenAI: o3",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-04-16 17:10:57 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "o3 is a well-rounded and powerful model across domains. It sets a new standard for math, science, coding, and visual reasoning tasks. It also excels at technical writing and instruction-following....",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openai/o3-deep-research",
"name": "OpenAI: o3 Deep Research",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-10-10 20:54:21 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 10.0,
"output_per_million": 40.0,
"cache_read_input_per_million": 2.5
}
}
},
"metadata": {
"description": "o3-deep-research is OpenAI's advanced model for deep research, designed to tackle complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/o3-mini",
"name": "OpenAI: o3 Mini",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-01-31 19:28:41 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.55
}
}
},
"metadata": {
"description": "OpenAI o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and coding. This model supports the `reasoning_effort` parameter, which can be set to...",
"architecture": {
"modality": "text+file->text",
"input_modalities": [
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openai/o3-mini-high",
"name": "OpenAI: o3 Mini High",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-02-12 15:03:31 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.55
}
}
},
"metadata": {
"description": "OpenAI o3-mini-high is the same model as [o3-mini](/openai/o3-mini) with reasoning_effort set to high. o3-mini is a cost-efficient language model optimized for STEM reasoning tasks, particularly excelling in science, mathematics, and...",
"architecture": {
"modality": "text+file->text",
"input_modalities": [
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openai/o3-pro",
"name": "OpenAI: o3 Pro",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-06-10 23:32:32 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"file",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 20.0,
"output_per_million": 80.0
}
}
},
"metadata": {
"description": "The o-series of models are trained with reinforcement learning to think before they answer and perform complex reasoning. The o3-pro model uses more compute to think harder and provide consistently...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"file",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openai/o4-mini",
"name": "o4 Mini",
"provider": "openrouter",
"family": "o-mini",
"created_at": "2025-04-16 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.28
}
}
},
"metadata": {
"description": "OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining strong multimodal and agentic capabilities. It supports tool use and demonstrates competitive reasoning...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-16",
"cost": {
"input": 1.1,
"output": 4.4,
"cache_read": 0.28
},
"limit": {
"context": 200000,
"output": 100000
},
"knowledge": "2024-06"
}
},
{
"id": "openai/o4-mini-deep-research",
"name": "OpenAI: o4 Mini Deep Research",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-10-10 20:54:02 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"file",
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "o4-mini-deep-research is OpenAI's faster, more affordable deep research model—ideal for tackling complex, multi-step research tasks.\n\nNote: This model always uses the 'web_search' tool which adds additional cost.",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "openai/o4-mini-high",
"name": "OpenAI: o4 Mini High",
"provider": "openrouter",
"family": "openai",
"created_at": "2025-04-16 17:23:32 UTC",
"context_window": 200000,
"max_output_tokens": 100000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.1,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.275
}
}
},
"metadata": {
"description": "OpenAI o4-mini-high is the same model as [o4-mini](/openai/o4-mini) with reasoning_effort set to high. OpenAI o4-mini is a compact reasoning model in the o-series, optimized for fast, cost-efficient performance while retaining...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "GPT",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 100000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "openrouter/auto",
"name": "Auto Router",
"provider": "openrouter",
"family": "openrouter",
"created_at": "2023-11-08 00:00:00 UTC",
"context_window": 2000000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"file",
"video"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {},
"metadata": {
"description": "\"Your prompt will be processed by a meta-model and routed to one of dozens of models (see below), optimizing for the best possible output. To see which model was used,...",
"architecture": {
"modality": "text+image+file+audio+video->text+image",
"input_modalities": [
"text",
"image",
"audio",
"file",
"video"
],
"output_modalities": [
"text",
"image"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": null,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_completion_tokens",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p",
"web_search_options"
]
}
},
{
"id": "openrouter/bodybuilder",
"name": "Body Builder (beta)",
"provider": "openrouter",
"family": "openrouter",
"created_at": "2025-12-05 03:00:53 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {},
"metadata": {
"description": "Transform your natural language requests into structured OpenRouter API request objects. Describe what you want to accomplish with AI models, and Body Builder will construct the appropriate API calls. Example:...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": null,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": []
}
},
{
"id": "openrouter/elephant-alpha",
"name": "Elephant (free)",
"provider": "openrouter",
"family": "elephant",
"created_at": "2026-04-13 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-13",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 262144,
"output": 32768
}
}
},
{
"id": "openrouter/free",
"name": "Free Models Router",
"provider": "openrouter",
"family": "openrouter",
"created_at": "2026-02-01 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {},
"metadata": {
"description": "The simplest way to get free inference. openrouter/free is a router that selects free models at random from the models available on OpenRouter. The router smartly filters for models that...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": null,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-01",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 200000,
"input": 200000,
"output": 8000
}
}
},
{
"id": "openrouter/owl-alpha",
"name": "Owl Alpha",
"provider": "openrouter",
"family": "openrouter",
"created_at": "2026-04-28 00:00:00 UTC",
"context_window": 1048756,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {},
"metadata": {
"description": "Owl Alpha is a high-performance foundation model designed for agentic workloads. Natively supports tool use, and long-context tasks, with strong performance in code generation, automated workflows, and complex instruction execution....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1048756,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-30",
"status": "alpha",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 1048756,
"output": 262144
}
}
},
{
"id": "openrouter/pareto-code",
"name": "Pareto Code Router",
"provider": "openrouter",
"family": "openrouter",
"created_at": "2026-04-21 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 200000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "The Pareto Router is a way to have OpenRouter always pick a strong coding model for your needs without committing to a specific one. You express a single `min_coding_score` preference...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": null,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-21",
"limit": {
"context": 200000,
"output": 200000
}
}
},
{
"id": "perplexity/sonar",
"name": "Perplexity: Sonar",
"provider": "openrouter",
"family": "perplexity",
"created_at": "2025-01-27 21:36:48 UTC",
"context_window": 127072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.0,
"output_per_million": 1.0
}
}
},
"metadata": {
"description": "Sonar is lightweight, affordable, fast, and simple to use — now featuring citations and the ability to customize sources. It is designed for companies seeking to integrate lightweight question-and-answer features...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 127072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"temperature",
"top_k",
"top_p",
"web_search_options"
]
}
},
{
"id": "perplexity/sonar-deep-research",
"name": "Perplexity: Sonar Deep Research",
"provider": "openrouter",
"family": "perplexity",
"created_at": "2025-03-07 01:34:06 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0,
"reasoning_output_per_million": 3.0
}
}
},
"metadata": {
"description": "Sonar Deep Research is a research-focused model designed for multi-step retrieval, synthesis, and reasoning across complex topics. It autonomously searches, reads, and evaluates sources, refining its approach as it gathers...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": "deepseek-r1"
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"temperature",
"top_k",
"top_p",
"web_search_options"
]
}
},
{
"id": "perplexity/sonar-pro",
"name": "Perplexity: Sonar Pro",
"provider": "openrouter",
"family": "perplexity",
"created_at": "2025-03-07 01:53:43 UTC",
"context_window": 200000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 15.0
}
}
},
"metadata": {
"description": "Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) For enterprises seeking more advanced capabilities, the Sonar Pro API can handle in-depth, multi-step queries with added extensibility, like...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 8000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"temperature",
"top_k",
"top_p",
"web_search_options"
]
}
},
{
"id": "perplexity/sonar-pro-search",
"name": "Perplexity: Sonar Pro Search",
"provider": "openrouter",
"family": "perplexity",
"created_at": "2025-10-30 19:59:26 UTC",
"context_window": 200000,
"max_output_tokens": 8000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 15.0
}
}
},
"metadata": {
"description": "Exclusively available on the OpenRouter API, Sonar Pro's new Pro Search mode is Perplexity's most advanced agentic search system. It is designed for deeper reasoning and analysis. Pricing is based...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 8000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"structured_outputs",
"temperature",
"top_k",
"top_p",
"web_search_options"
]
}
},
{
"id": "perplexity/sonar-reasoning-pro",
"name": "Perplexity: Sonar Reasoning Pro",
"provider": "openrouter",
"family": "perplexity",
"created_at": "2025-03-07 02:08:28 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 8.0
}
}
},
"metadata": {
"description": "Note: Sonar Pro pricing includes Perplexity search pricing. See [details here](https://docs.perplexity.ai/guides/pricing#detailed-pricing-breakdown-for-sonar-reasoning-pro-and-sonar-pro) Sonar Reasoning Pro is a premier reasoning model powered by DeepSeek R1 with Chain of Thought (CoT). Designed for...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": "deepseek-r1"
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"temperature",
"top_k",
"top_p",
"web_search_options"
]
}
},
{
"id": "poolside/laguna-m.1:free",
"name": "Laguna M.1",
"provider": "openrouter",
"family": "poolside",
"created_at": "2026-04-28 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Laguna M.1 is the flagship coding agent model from [Poolside](https://poolside.ai), optimized for complex software engineering tasks. Designed for agentic coding workflows, it supports tool calling and reasoning, with a 128K...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"temperature",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-28",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0,
"output": 0,
"cache_read": 0,
"cache_write": 0
},
"limit": {
"context": 131072,
"output": 8192
}
}
},
{
"id": "poolside/laguna-xs.2:free",
"name": "Laguna XS.2",
"provider": "openrouter",
"family": "poolside",
"created_at": "2026-04-28 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming"
],
"pricing": {},
"metadata": {
"description": "Laguna XS.2 is the second-generation model in the XS size class from [Poolside](https://poolside.ai), their efficient coding agent series. It combines tool calling and reasoning capabilities with a compact footprint, offering...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"temperature",
"tool_choice",
"tools"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-28",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0,
"output": 0,
"cache_read": 0,
"cache_write": 0
},
"limit": {
"context": 131072,
"output": 8192
}
}
},
{
"id": "prime-intellect/intellect-3",
"name": "Intellect 3",
"provider": "openrouter",
"family": "glm",
"created_at": "2025-01-15 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 1.1
}
}
},
"metadata": {
"description": "INTELLECT-3 is a 106B-parameter Mixture-of-Experts model (12B active) post-trained from GLM-4.5-Air-Base using supervised fine-tuning (SFT) followed by large-scale reinforcement learning (RL). It offers state-of-the-art performance for its size across math,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-01-15",
"cost": {
"input": 0.2,
"output": 1.1
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "qwen/qwen-2.5-72b-instruct",
"name": "Qwen2.5 72B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2024-09-19 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.36,
"output_per_million": 0.39999999999999997
}
}
},
"metadata": {
"description": "Qwen2.5 72B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen-2.5-7b-instruct",
"name": "Qwen: Qwen2.5 7B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2024-10-16 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.09999999999999999
}
}
},
"metadata": {
"description": "Qwen2.5 7B is the latest series of Qwen large language models. Qwen2.5 brings the following improvements upon Qwen2: - Significantly more knowledge and has greatly improved capabilities in coding and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen-2.5-coder-32b-instruct",
"name": "Qwen2.5 Coder 32B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2024-11-11 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.66,
"output_per_million": 1.0
}
}
},
"metadata": {
"description": "Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). Qwen2.5-Coder brings the following improvements upon CodeQwen1.5: - Significantly improvements in **code generation**, **code reasoning**...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2024-11-11",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "qwen/qwen-3.6-27b",
"name": "Qwen3.6 27B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-22 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 81920,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.195,
"output_per_million": 1.56
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-22",
"cost": {
"input": 0.195,
"output": 1.56
},
"limit": {
"context": 262144,
"output": 81920
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen-max",
"name": "Qwen: Qwen-Max ",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-02-01 09:31:29 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.04,
"output_per_million": 4.16,
"cache_read_input_per_million": 0.20800000000000002
}
}
},
"metadata": {
"description": "Qwen-Max, based on Qwen2.5, provides the best inference performance among [Qwen models](/qwen), especially for complex multi-step tasks. It's a large-scale MoE model that has been pretrained on over 20 trillion...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen-plus",
"name": "Qwen: Qwen-Plus",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-02-01 11:37:20 UTC",
"context_window": 1000000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.26,
"output_per_million": 0.78,
"cache_read_input_per_million": 0.052000000000000005
}
}
},
"metadata": {
"description": "Qwen-Plus, based on the Qwen2.5 foundation model, is a 131K context model with a balanced performance, speed, and cost combination.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen-plus-2025-07-28",
"name": "Qwen: Qwen Plus 0728",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-08 16:06:39 UTC",
"context_window": 1000000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.26,
"output_per_million": 0.78
}
}
},
"metadata": {
"description": "Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen-plus-2025-07-28:thinking",
"name": "Qwen: Qwen Plus 0728 (thinking)",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-08 16:06:39 UTC",
"context_window": 1000000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.26,
"output_per_million": 0.78
}
}
},
"metadata": {
"description": "Qwen Plus 0728, based on the Qwen3 foundation model, is a 1 million context hybrid reasoning model with a balanced performance, speed, and cost combination.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen-turbo",
"name": "Qwen: Qwen-Turbo",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-02-01 11:56:14 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.0325,
"output_per_million": 0.13,
"cache_read_input_per_million": 0.006500000000000001
}
}
},
"metadata": {
"description": "Qwen-Turbo, based on Qwen2.5, is a 1M context model that provides fast speed and low cost, suitable for simple tasks.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen-vl-max",
"name": "Qwen: Qwen VL Max",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-02-01 18:25:04 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.52,
"output_per_million": 2.08
}
}
},
"metadata": {
"description": "Qwen VL Max is a visual understanding model with 7500 tokens context length. It excels in delivering optimal performance for a broader spectrum of complex tasks.\n",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen-vl-plus",
"name": "Qwen: Qwen VL Plus",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-02-05 04:54:15 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1365,
"output_per_million": 0.40950000000000003,
"cache_read_input_per_million": 0.027299999999999998
}
}
},
"metadata": {
"description": "Qwen's Enhanced Large Visual Language Model. Significantly upgraded for detailed recognition capabilities and text recognition abilities, supporting ultra-high pixel resolutions up to millions of pixels and extreme aspect ratios for...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"top_p"
]
}
},
{
"id": "qwen/qwen2.5-vl-72b-instruct",
"name": "Qwen2.5 VL 72B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-02-01 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"structured_output",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 0.75
}
}
},
"metadata": {
"description": "Qwen2.5-VL is proficient in recognizing common objects such as flowers, birds, fish, and insects. It is also highly capable of analyzing texts, charts, icons, graphics, and layouts within images.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 32000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-02-01",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 32768,
"output": 8192
},
"knowledge": "2024-10"
}
},
{
"id": "qwen/qwen3-14b",
"name": "Qwen: Qwen3 14B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-04-28 21:41:18 UTC",
"context_window": 40960,
"max_output_tokens": 40960,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.06,
"output_per_million": 0.24
}
}
},
"metadata": {
"description": "Qwen3-14B is a dense 14.8B parameter causal language model from the Qwen3 series, designed for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": "qwen3"
},
"top_provider": {
"context_length": 40960,
"max_completion_tokens": 40960,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3-235b-a22b",
"name": "Qwen: Qwen3 235B A22B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-04-28 21:29:17 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.45499999999999996,
"output_per_million": 1.8199999999999998
}
}
},
"metadata": {
"description": "Qwen3-235B-A22B is a 235B parameter mixture-of-experts (MoE) model developed by Qwen, activating 22B parameters per forward pass. It supports seamless switching between a \"thinking\" mode for complex reasoning, math, and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": "qwen3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3-235b-a22b-07-25",
"name": "Qwen3 235B A22B Instruct 2507",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-04-28 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.85
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-21",
"cost": {
"input": 0.15,
"output": 0.85
},
"limit": {
"context": 262144,
"output": 131072
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-235b-a22b-2507",
"name": "Qwen: Qwen3 235B A22B Instruct 2507",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-21 17:39:15 UTC",
"context_window": 262144,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.071,
"output_per_million": 0.09999999999999999
}
}
},
"metadata": {
"description": "Qwen3-235B-A22B-Instruct-2507 is a multilingual, instruction-tuned mixture-of-experts language model based on the Qwen3-235B architecture, with 22B active parameters per forward pass. It is optimized for general-purpose text generation, including instruction following,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3-235b-a22b-thinking-2507",
"name": "Qwen3 235B A22B Thinking 2507",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-25 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 81920,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.078,
"output_per_million": 0.312
}
}
},
"metadata": {
"description": "Qwen3-235B-A22B-Thinking-2507 is a high-performance, open-weight Mixture-of-Experts (MoE) language model optimized for complex reasoning tasks. It activates 22B of its 235B parameters per forward pass and natively supports up to 262,144...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": "qwen3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-25",
"cost": {
"input": 0.078,
"output": 0.312
},
"limit": {
"context": 262144,
"output": 81920
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-30b-a3b",
"name": "Qwen: Qwen3 30B A3B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-04-28 22:16:44 UTC",
"context_window": 40960,
"max_output_tokens": 20000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09,
"output_per_million": 0.44999999999999996
}
}
},
"metadata": {
"description": "Qwen3, the latest generation in the Qwen large language model series, features both dense and mixture-of-experts (MoE) architectures to excel in reasoning, multilingual support, and advanced agent tasks. Its unique...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": "qwen3"
},
"top_provider": {
"context_length": 40960,
"max_completion_tokens": 20000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3-30b-a3b-instruct-2507",
"name": "Qwen3 30B A3B Instruct 2507",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-29 00:00:00 UTC",
"context_window": 262000,
"max_output_tokens": 262000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.8
}
}
},
"metadata": {
"description": "Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. It operates in non-thinking mode and is designed for high-quality instruction following, multilingual understanding, and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-29",
"cost": {
"input": 0.2,
"output": 0.8
},
"limit": {
"context": 262000,
"output": 262000
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-30b-a3b-thinking-2507",
"name": "Qwen3 30B A3B Thinking 2507",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-29 00:00:00 UTC",
"context_window": 262000,
"max_output_tokens": 262000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.8
}
}
},
"metadata": {
"description": "Qwen3-30B-A3B-Thinking-2507 is a 30B parameter Mixture-of-Experts reasoning model optimized for complex tasks requiring extended multi-step thinking. The model is designed specifically for “thinking mode,” where internal reasoning traces are separated...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-29",
"cost": {
"input": 0.2,
"output": 0.8
},
"limit": {
"context": 262000,
"output": 262000
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-32b",
"name": "Qwen: Qwen3 32B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-04-28 21:32:25 UTC",
"context_window": 40960,
"max_output_tokens": 40960,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.08,
"output_per_million": 0.24,
"cache_read_input_per_million": 0.04
}
}
},
"metadata": {
"description": "Qwen3-32B is a dense 32.8B parameter causal language model from the Qwen3 series, optimized for both complex reasoning and efficient dialogue. It supports seamless switching between a \"thinking\" mode for...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": "qwen3"
},
"top_provider": {
"context_length": 40960,
"max_completion_tokens": 40960,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-8b",
"name": "Qwen: Qwen3 8B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-04-28 21:43:52 UTC",
"context_window": 40960,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.049999999999999996,
"output_per_million": 0.39999999999999997,
"cache_read_input_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "Qwen3-8B is a dense 8.2B parameter causal language model from the Qwen3 series, designed for both reasoning-heavy tasks and efficient dialogue. It supports seamless switching between \"thinking\" mode for math,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": "qwen3"
},
"top_provider": {
"context_length": 40960,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-coder",
"name": "Qwen3 Coder",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-23 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 66536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.2
}
}
},
"metadata": {
"description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-23",
"cost": {
"input": 0.3,
"output": 1.2
},
"limit": {
"context": 262144,
"output": 66536
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-coder-30b-a3b-instruct",
"name": "Qwen3 Coder 30B A3B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-31 00:00:00 UTC",
"context_window": 160000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.27
}
}
},
"metadata": {
"description": "Qwen3-Coder-30B-A3B-Instruct is a 30.5B parameter Mixture-of-Experts (MoE) model with 128 experts (8 active per forward pass), designed for advanced code generation, repository-scale understanding, and agentic tool use. Built on the...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 160000,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-31",
"cost": {
"input": 0.07,
"output": 0.27
},
"limit": {
"context": 160000,
"output": 65536
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-coder-flash",
"name": "Qwen3 Coder Flash",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-23 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 66536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.5
}
}
},
"metadata": {
"description": "Qwen3 Coder Flash is Alibaba's fast and cost efficient version of their proprietary Qwen3 Coder Plus. It is a powerful coding agent model specializing in autonomous programming via tool calling...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-23",
"cost": {
"input": 0.3,
"output": 1.5
},
"limit": {
"context": 128000,
"output": 66536
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-coder-next",
"name": "Qwen: Qwen3 Coder Next",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-04 00:15:01 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.11,
"output_per_million": 0.7999999999999999,
"cache_read_input_per_million": 0.07
}
}
},
"metadata": {
"description": "Qwen3-Coder-Next is an open-weight causal language model optimized for coding agents and local development workflows. It uses a sparse MoE design with 80B total parameters and only 3B activated per...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-coder-plus",
"name": "Qwen: Qwen3 Coder Plus",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-23 21:25:07 UTC",
"context_window": 1000000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.65,
"output_per_million": 3.25,
"cache_read_input_per_million": 0.13
}
}
},
"metadata": {
"description": "Qwen3 Coder Plus is Alibaba's proprietary version of the Open Source Qwen3 Coder 480B A35B. It is a powerful coding agent model specializing in autonomous programming via tool calling and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3-coder:exacto",
"name": "Qwen3 Coder (exacto)",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-23 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.38,
"output_per_million": 1.53
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-23",
"cost": {
"input": 0.38,
"output": 1.53
},
"limit": {
"context": 131072,
"output": 32768
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-coder:free",
"name": "Qwen: Qwen3 Coder 480B A35B (free)",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-07-23 00:29:06 UTC",
"context_window": 262000,
"max_output_tokens": 262000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"description": "Qwen3-Coder-480B-A35B-Instruct is a Mixture-of-Experts (MoE) code generation model developed by the Qwen team. It is optimized for agentic coding tasks such as function calling, tool use, and long-context reasoning over...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262000,
"max_completion_tokens": 262000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-max",
"name": "Qwen3 Max",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-05 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.2,
"output_per_million": 6
}
}
},
"metadata": {
"description": "Qwen3-Max is an updated release built on the Qwen3 series, offering major improvements in reasoning, instruction following, multilingual support, and long-tail knowledge coverage compared to the January 2025 version. It...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-05",
"cost": {
"input": 1.2,
"output": 6
},
"limit": {
"context": 262144,
"output": 32768
}
}
},
{
"id": "qwen/qwen3-max-thinking",
"name": "Qwen: Qwen3 Max Thinking",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-09 21:18:21 UTC",
"context_window": 262144,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.78,
"output_per_million": 3.9
}
}
},
"metadata": {
"description": "Qwen3-Max-Thinking is the flagship reasoning model in the Qwen3 series, designed for high-stakes cognitive tasks that require deep, multi-step reasoning. By significantly scaling model capacity and reinforcement learning compute, it...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3-next-80b-a3b-instruct",
"name": "Qwen3 Next 80B A3B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-11 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 1.4
}
}
},
"metadata": {
"description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-11",
"cost": {
"input": 0.14,
"output": 1.4
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-next-80b-a3b-instruct:free",
"name": "Qwen: Qwen3 Next 80B A3B Instruct (free)",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-11 17:36:53 UTC",
"context_window": 262144,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"description": "Qwen3-Next-80B-A3B-Instruct is an instruction-tuned chat model in the Qwen3-Next series optimized for fast, stable responses without “thinking” traces. It targets complex tasks across reasoning, code generation, knowledge QA, and multilingual...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-next-80b-a3b-thinking",
"name": "Qwen3 Next 80B A3B Thinking",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-11 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 1.4
}
}
},
"metadata": {
"description": "Qwen3-Next-80B-A3B-Thinking is a reasoning-first chat model in the Qwen3-Next line that outputs structured “thinking” traces by default. It’s designed for hard multi-step problems; math proofs, code synthesis/debugging, logic, and agentic...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-11",
"cost": {
"input": 0.14,
"output": 1.4
},
"limit": {
"context": 262144,
"output": 262144
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3-vl-235b-a22b-instruct",
"name": "Qwen: Qwen3 VL 235B A22B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-23 23:04:47 UTC",
"context_window": 262144,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.19999999999999998,
"output_per_million": 0.88,
"cache_read_input_per_million": 0.11
}
}
},
"metadata": {
"description": "Qwen3-VL-235B-A22B Instruct is an open-weight multimodal model that unifies strong text generation with visual understanding across images and video. The Instruct model targets general vision-language use (VQA, document parsing, chart/table...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-vl-235b-a22b-thinking",
"name": "Qwen: Qwen3 VL 235B A22B Thinking",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-09-23 23:04:50 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.26,
"output_per_million": 2.6
}
}
},
"metadata": {
"description": "Qwen3-VL-235B-A22B Thinking is a multimodal model that unifies strong text generation with visual understanding across images and video. The Thinking model is optimized for multimodal reasoning in STEM and math....",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-vl-30b-a3b-instruct",
"name": "Qwen: Qwen3 VL 30B A3B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-10-06 23:47:56 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.13,
"output_per_million": 0.52
}
}
},
"metadata": {
"description": "Qwen3-VL-30B-A3B-Instruct is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Instruct variant optimizes instruction-following for general multimodal tasks. It excels in perception...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-vl-30b-a3b-thinking",
"name": "Qwen: Qwen3 VL 30B A3B Thinking",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-10-06 23:47:59 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.13,
"output_per_million": 1.56
}
}
},
"metadata": {
"description": "Qwen3-VL-30B-A3B-Thinking is a multimodal model that unifies strong text generation with visual understanding for images and videos. Its Thinking variant enhances reasoning in STEM, math, and complex tasks. It excels...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-vl-32b-instruct",
"name": "Qwen: Qwen3 VL 32B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-10-23 14:55:32 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.10400000000000001,
"output_per_million": 0.41600000000000004
}
}
},
"metadata": {
"description": "Qwen3-VL-32B-Instruct is a large-scale multimodal vision-language model designed for high-precision understanding and reasoning across text, images, and video. With 32 billion parameters, it combines deep visual perception with advanced text...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3-vl-8b-instruct",
"name": "Qwen: Qwen3 VL 8B Instruct",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-10-14 17:35:08 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.08,
"output_per_million": 0.5
}
}
},
"metadata": {
"description": "Qwen3-VL-8B-Instruct is a multimodal vision-language model from the Qwen3-VL series, built for high-fidelity understanding and reasoning across text, images, and video. It features improved multimodal fusion with Interleaved-MRoPE for long-horizon...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3-vl-8b-thinking",
"name": "Qwen: Qwen3 VL 8B Thinking",
"provider": "openrouter",
"family": "qwen",
"created_at": "2025-10-14 17:42:26 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.117,
"output_per_million": 1.365
}
}
},
"metadata": {
"description": "Qwen3-VL-8B-Thinking is the reasoning-optimized variant of the Qwen3-VL-8B multimodal model, designed for advanced visual and textual reasoning across complex scenes, documents, and temporal sequences. It integrates enhanced multimodal alignment and...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3.5-122b-a10b",
"name": "Qwen: Qwen3.5-122B-A10B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-25 21:09:49 UTC",
"context_window": 262144,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.26,
"output_per_million": 2.08
}
}
},
"metadata": {
"description": "The Qwen3.5 122B-A10B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. In terms of...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3.5-27b",
"name": "Qwen: Qwen3.5-27B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-25 21:10:10 UTC",
"context_window": 262144,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.195,
"output_per_million": 1.56
}
}
},
"metadata": {
"description": "The Qwen3.5 27B native vision-language Dense model incorporates a linear attention mechanism, delivering fast response times while balancing inference speed and performance. Its overall capabilities are comparable to those of...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3.5-35b-a3b",
"name": "Qwen: Qwen3.5-35B-A3B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-25 21:10:22 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 1.0,
"cache_read_input_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "The Qwen3.5 Series 35B-A3B is a native vision-language model designed with a hybrid architecture that integrates linear attention mechanisms and a sparse mixture-of-experts model, achieving higher inference efficiency. Its overall...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3.5-397b-a17b",
"name": "Qwen3.5 397B A17B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-16 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 3.6
}
}
},
"metadata": {
"description": "The Qwen3.5 series 397B-A17B native vision-language model is built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. It delivers...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-16",
"cost": {
"input": 0.6,
"output": 3.6
},
"limit": {
"context": 262144,
"output": 65536
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3.5-9b",
"name": "Qwen: Qwen3.5-9B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-03-10 14:19:56 UTC",
"context_window": 262144,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.15
}
}
},
"metadata": {
"description": "Qwen3.5-9B is a multimodal foundation model from the Qwen3.5 family, designed to deliver strong reasoning, coding, and visual understanding in an efficient 9B-parameter architecture. It uses a unified vision-language design...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3.5-flash-02-23",
"name": "Qwen: Qwen3.5-Flash",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-25 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.065,
"output_per_million": 0.26
}
}
},
"metadata": {
"description": "The Qwen3.5 native vision-language Flash models are built on a hybrid architecture that integrates a linear attention mechanism with a sparse mixture-of-experts model, achieving higher inference efficiency. Compared to the...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-25",
"cost": {
"input": 0.065,
"output": 0.26
},
"limit": {
"context": 1000000,
"output": 65536
}
}
},
{
"id": "qwen/qwen3.5-plus-02-15",
"name": "Qwen3.5 Plus 2026-02-15",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-02-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2.4
}
}
},
"metadata": {
"description": "The Qwen3.5 native vision-language series Plus models are built on a hybrid architecture that integrates linear attention mechanisms with sparse mixture-of-experts models, achieving higher inference efficiency. In a variety of...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-16",
"cost": {
"input": 0.4,
"output": 2.4
},
"limit": {
"context": 1000000,
"output": 65536
},
"knowledge": "2025-04"
}
},
{
"id": "qwen/qwen3.5-plus-20260420",
"name": "Qwen: Qwen3.5 Plus 2026-04-20",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-27 03:42:48 UTC",
"context_window": 1000000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.39999999999999997,
"output_per_million": 2.4
}
}
},
"metadata": {
"description": "Qwen3.5 Plus (April 2026) is a large-scale multimodal language model from Alibaba. It accepts text, image, and video input and produces text output, with a 1M token context window. This...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3.6-27b",
"name": "Qwen: Qwen3.6 27B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-27 01:57:44 UTC",
"context_window": 262144,
"max_output_tokens": 81920,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.32,
"output_per_million": 3.1999999999999997
}
}
},
"metadata": {
"description": "Qwen3.6 27B is a dense 27-billion-parameter language model from the Qwen Team at Alibaba, released in April 2026. It features hybrid multimodal capabilities — accepting text, image, and video inputs...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 81920,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3.6-35b-a3b",
"name": "Qwen: Qwen3.6 35B A3B",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-27 03:24:15 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 1.0,
"cache_read_input_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "Qwen3.6-35B-A3B is an open-weight multimodal model from Alibaba Cloud with 35 billion total parameters and 3 billion active parameters per token. It uses a hybrid sparse mixture-of-experts architecture combining Gated...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "qwen/qwen3.6-flash",
"name": "Qwen: Qwen3.6 Flash",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-27 03:42:42 UTC",
"context_window": 1000000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.5
}
}
},
"metadata": {
"description": "Qwen3.6 Flash is a fast, efficient language model from Alibaba's Qwen 3.6 series. It supports text, image, and video input with a 1M token context window. Tiered pricing kicks in...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "qwen/qwen3.6-max-preview",
"name": "Qwen: Qwen3.6 Max Preview",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-27 03:24:02 UTC",
"context_window": 262144,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.04,
"output_per_million": 6.24
}
}
},
"metadata": {
"description": "Qwen3.6-Max-Preview is a proprietary frontier model from Alibaba Cloud built on a sparse mixture-of-experts architecture with approximately 1 trillion total parameters. It is optimized for agentic coding, tool use, and...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "qwen/qwen3.6-plus",
"name": "Qwen3.6 Plus",
"provider": "openrouter",
"family": "qwen",
"created_at": "2026-04-02 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.325,
"output_per_million": 1.95
}
}
},
"metadata": {
"description": "Qwen 3.6 Plus builds on a hybrid architecture that combines efficient linear attention with sparse mixture-of-experts routing, enabling strong scalability and high-performance inference. Compared to the 3.5 series, it delivers...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"text",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen3",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-02",
"cost": {
"input": 0.325,
"output": 1.95
},
"limit": {
"context": 1000000,
"output": 65536
},
"knowledge": "2025-04"
}
},
{
"id": "rekaai/reka-edge",
"name": "Reka Edge",
"provider": "openrouter",
"family": "rekaai",
"created_at": "2026-03-20 17:16:05 UTC",
"context_window": 16384,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.09999999999999999
}
}
},
"metadata": {
"description": "Reka Edge is an extremely efficient 7B multimodal vision-language model that accepts image/video+text inputs and generates text outputs. This model is optimized specifically to deliver industry-leading performance in image understanding,...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 16384,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "rekaai/reka-flash-3",
"name": "Reka Flash 3",
"provider": "openrouter",
"family": "rekaai",
"created_at": "2025-03-12 20:53:33 UTC",
"context_window": 65536,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Reka Flash 3 is a general-purpose, instruction-tuned large language model with 21 billion parameters, developed by Reka. It excels at general chat, coding tasks, instruction-following, and function calling. Featuring a...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "relace/relace-apply-3",
"name": "Relace: Relace Apply 3",
"provider": "openrouter",
"family": "relace",
"created_at": "2025-09-26 12:59:32 UTC",
"context_window": 256000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.85,
"output_per_million": 1.25
}
}
},
"metadata": {
"description": "Relace Apply 3 is a specialized code-patching LLM that merges AI-suggested edits straight into your source files. It can apply updates from GPT-4o, Claude, and others into your files at...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"seed",
"stop"
]
}
},
{
"id": "relace/relace-search",
"name": "Relace: Relace Search",
"provider": "openrouter",
"family": "relace",
"created_at": "2025-12-08 17:06:00 UTC",
"context_window": 256000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.0,
"output_per_million": 3.0
}
}
},
"metadata": {
"description": "The relace-search model uses 4-12 `view_file` and `grep` tools in parallel to explore a codebase and return relevant files to the user request. In contrast to RAG, relace-search performs agentic...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "sao10k/l3-euryale-70b",
"name": "Sao10k: Llama 3 Euryale 70B v2.1",
"provider": "openrouter",
"family": "sao10k",
"created_at": "2024-06-18 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.48,
"output_per_million": 1.48
}
}
},
"metadata": {
"description": "Euryale 70B v2.1 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). - Better prompt adherence. - Better anatomy / spatial awareness. - Adapts much better to unique and custom...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 8192,
"max_completion_tokens": 8192,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "sao10k/l3-lunaris-8b",
"name": "Sao10K: Llama 3 8B Lunaris",
"provider": "openrouter",
"family": "sao10k",
"created_at": "2024-08-13 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.04,
"output_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "Lunaris 8B is a versatile generalist and roleplaying model based on Llama 3. It's a strategic merge of multiple models, designed to balance creativity with improved logic and general knowledge....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 8192,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "sao10k/l3.1-70b-hanami-x1",
"name": "Sao10K: Llama 3.1 70B Hanami x1",
"provider": "openrouter",
"family": "sao10k",
"created_at": "2025-01-08 02:20:54 UTC",
"context_window": 16000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 3.0
}
}
},
"metadata": {
"description": "This is [Sao10K](/sao10k)'s experiment over [Euryale v2.2](/sao10k/l3.1-euryale-70b).",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": null
},
"top_provider": {
"context_length": 16000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "sao10k/l3.1-euryale-70b",
"name": "Sao10K: Llama 3.1 Euryale 70B v2.2",
"provider": "openrouter",
"family": "sao10k",
"created_at": "2024-08-28 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.85,
"output_per_million": 0.85
}
}
},
"metadata": {
"description": "Euryale L3.1 70B v2.2 is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.1](/models/sao10k/l3-euryale-70b).",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "sao10k/l3.3-euryale-70b",
"name": "Sao10K: Llama 3.3 Euryale 70B",
"provider": "openrouter",
"family": "sao10k",
"created_at": "2024-12-18 15:32:08 UTC",
"context_window": 131072,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.65,
"output_per_million": 0.75
}
}
},
"metadata": {
"description": "Euryale L3.3 70B is a model focused on creative roleplay from [Sao10k](https://ko-fi.com/sao10k). It is the successor of [Euryale L3 70B v2.2](/models/sao10k/l3-euryale-70b).",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama3",
"instruct_type": "llama3"
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "sourceful/riverflow-v2-fast-preview",
"name": "Riverflow V2 Fast Preview",
"provider": "openrouter",
"family": "sourceful",
"created_at": "2025-12-08 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 8192
},
"knowledge": "2025-06"
}
},
{
"id": "sourceful/riverflow-v2-max-preview",
"name": "Riverflow V2 Max Preview",
"provider": "openrouter",
"family": "sourceful",
"created_at": "2025-12-08 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 8192
},
"knowledge": "2025-06"
}
},
{
"id": "sourceful/riverflow-v2-standard-preview",
"name": "Riverflow V2 Standard Preview",
"provider": "openrouter",
"family": "sourceful",
"created_at": "2025-12-08 00:00:00 UTC",
"context_window": 8192,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"image"
]
},
"capabilities": [
"vision"
],
"pricing": {},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 8192,
"output": 8192
},
"knowledge": "2025-06"
}
},
{
"id": "stepfun/step-3.5-flash",
"name": "Step 3.5 Flash",
"provider": "openrouter",
"family": "step",
"created_at": "2026-01-29 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 256000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.3,
"cache_read_input_per_million": 0.02
}
}
},
"metadata": {
"description": "Step 3.5 Flash is StepFun's most capable open-source foundation model. Built on a sparse Mixture of Experts (MoE) architecture, it selectively activates only 11B of its 196B parameters per token....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-29",
"cost": {
"input": 0.1,
"output": 0.3,
"cache_read": 0.02
},
"limit": {
"context": 256000,
"output": 256000
},
"knowledge": "2025-01"
}
},
{
"id": "switchpoint/router",
"name": "Switchpoint Router",
"provider": "openrouter",
"family": "switchpoint",
"created_at": "2025-07-11 22:28:19 UTC",
"context_window": 131072,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.85,
"output_per_million": 3.4
}
}
},
"metadata": {
"description": "Switchpoint AI's router instantly analyzes your request and directs it to the optimal AI from an ever-evolving library. As the world of LLMs advances, our router gets smarter, ensuring you...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "tencent/hunyuan-a13b-instruct",
"name": "Tencent: Hunyuan A13B Instruct",
"provider": "openrouter",
"family": "tencent",
"created_at": "2025-07-08 15:14:24 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.14,
"output_per_million": 0.5700000000000001
}
}
},
"metadata": {
"description": "Hunyuan-A13B is a 13B active parameter Mixture-of-Experts (MoE) language model developed by Tencent, with a total parameter count of 80B and support for reasoning via Chain-of-Thought. It offers competitive benchmark...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"structured_outputs",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "tencent/hy3-preview:free",
"name": "Tencent: Hy3 preview (free)",
"provider": "openrouter",
"family": "tencent",
"created_at": "2026-04-22 17:15:50 UTC",
"context_window": 262144,
"max_output_tokens": 262144,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"description": "Hy3 preview is a high-efficiency Mixture-of-Experts model from Tencent designed for agentic workflows and production use. It supports configurable reasoning levels across disabled, low, and high modes, allowing it to...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 262144,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "thedrummer/cydonia-24b-v4.1",
"name": "TheDrummer: Cydonia 24B V4.1",
"provider": "openrouter",
"family": "thedrummer",
"created_at": "2025-09-27 00:11:18 UTC",
"context_window": 131072,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.5,
"cache_read_input_per_million": 0.15
}
}
},
"metadata": {
"description": "Uncensored and creative writing model based on Mistral Small 3.2 24B with good recall, prompt adherence, and intelligence.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "thedrummer/rocinante-12b",
"name": "TheDrummer: Rocinante 12B",
"provider": "openrouter",
"family": "thedrummer",
"created_at": "2024-09-30 00:00:00 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.16999999999999998,
"output_per_million": 0.43
}
}
},
"metadata": {
"description": "Rocinante 12B is designed for engaging storytelling and rich prose. Early testers have reported: - Expanded vocabulary with unique and expressive word choices - Enhanced creativity for vivid narratives -...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Qwen",
"instruct_type": "chatml"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "thedrummer/skyfall-36b-v2",
"name": "TheDrummer: Skyfall 36B V2",
"provider": "openrouter",
"family": "thedrummer",
"created_at": "2025-03-10 19:56:06 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.55,
"output_per_million": 0.7999999999999999,
"cache_read_input_per_million": 0.25
}
}
},
"metadata": {
"description": "Skyfall 36B v2 is an enhanced iteration of Mistral Small 2501, specifically fine-tuned for improved creativity, nuanced writing, role-playing, and coherent storytelling.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"seed",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "thedrummer/unslopnemo-12b",
"name": "TheDrummer: UnslopNemo 12B",
"provider": "openrouter",
"family": "thedrummer",
"created_at": "2024-11-08 22:04:08 UTC",
"context_window": 32768,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.39999999999999997,
"output_per_million": 0.39999999999999997
}
}
},
"metadata": {
"description": "UnslopNemo v4.1 is the latest addition from the creator of Rocinante, designed for adventure writing and role-play scenarios.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Mistral",
"instruct_type": "mistral"
},
"top_provider": {
"context_length": 32768,
"max_completion_tokens": 32768,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logprobs",
"max_tokens",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "tngtech/deepseek-r1t2-chimera",
"name": "TNG: DeepSeek R1T2 Chimera",
"provider": "openrouter",
"family": "tngtech",
"created_at": "2025-07-08 15:03:05 UTC",
"context_window": 163840,
"max_output_tokens": 163840,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 1.1,
"cache_read_input_per_million": 0.15
}
}
},
"metadata": {
"description": "DeepSeek-TNG-R1T2-Chimera is the second-generation Chimera model from TNG Tech. It is a 671 B-parameter mixture-of-experts text-generation model assembled from DeepSeek-AI’s R1-0528, R1, and V3-0324 checkpoints with an Assembly-of-Experts merge. The...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "DeepSeek",
"instruct_type": null
},
"top_provider": {
"context_length": 163840,
"max_completion_tokens": 163840,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "undi95/remm-slerp-l2-13b",
"name": "ReMM SLERP 13B",
"provider": "openrouter",
"family": "undi95",
"created_at": "2023-07-22 00:00:00 UTC",
"context_window": 6144,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.44999999999999996,
"output_per_million": 0.65
}
}
},
"metadata": {
"description": "A recreation trial of the original MythoMax-L2-B13 but with updated models. #merge",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Llama2",
"instruct_type": "alpaca"
},
"top_provider": {
"context_length": 6144,
"max_completion_tokens": 4096,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"top_a",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "upstage/solar-pro-3",
"name": "Upstage: Solar Pro 3",
"provider": "openrouter",
"family": "upstage",
"created_at": "2026-01-27 02:33:20 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.015
}
}
},
"metadata": {
"description": "Solar Pro 3 is Upstage's powerful Mixture-of-Experts (MoE) language model. With 102B total parameters and 12B active parameters per forward pass, it delivers exceptional performance while maintaining computational efficiency. Optimized...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"structured_outputs",
"temperature",
"tool_choice",
"tools"
]
}
},
{
"id": "writer/palmyra-x5",
"name": "Writer: Palmyra X5",
"provider": "openrouter",
"family": "writer",
"created_at": "2026-01-21 13:57:03 UTC",
"context_window": 1040000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 6.0
}
}
},
"metadata": {
"description": "Palmyra X5 is Writer's most advanced model, purpose-built for building and scaling AI agents across the enterprise. It delivers industry-leading speed and efficiency on context windows up to 1 million...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1040000,
"max_completion_tokens": 8192,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"stop",
"temperature",
"top_k",
"top_p"
]
}
},
{
"id": "x-ai/grok-3",
"name": "Grok 3",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-02-17 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.75,
"cache_write_input_per_million": 15
}
}
},
"metadata": {
"description": "Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-02-17",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.75,
"cache_write": 15
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-11"
}
},
{
"id": "x-ai/grok-3-beta",
"name": "Grok 3 Beta",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-02-17 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.75,
"cache_write_input_per_million": 15
}
}
},
"metadata": {
"description": "Grok 3 is the latest model from xAI. It's their flagship model that excels at enterprise use cases like data extraction, coding, and text summarization. Possesses deep domain knowledge in...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"logprobs",
"max_tokens",
"presence_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-02-17",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.75,
"cache_write": 15
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-11"
}
},
{
"id": "x-ai/grok-3-mini",
"name": "Grok 3 Mini",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-02-17 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.5,
"cache_read_input_per_million": 0.075,
"cache_write_input_per_million": 0.5
}
}
},
"metadata": {
"description": "A lightweight model that thinks before responding. Fast, smart, and great for logic-based tasks that do not require deep domain knowledge. The raw thinking traces are accessible.",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-02-17",
"cost": {
"input": 0.3,
"output": 0.5,
"cache_read": 0.075,
"cache_write": 0.5
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-11"
}
},
{
"id": "x-ai/grok-3-mini-beta",
"name": "Grok 3 Mini Beta",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-02-17 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.5,
"cache_read_input_per_million": 0.075,
"cache_write_input_per_million": 0.5
}
}
},
"metadata": {
"description": "Grok 3 Mini is a lightweight, smaller thinking model. Unlike traditional models that generate answers immediately, Grok 3 Mini thinks before responding. It’s ideal for reasoning-heavy tasks that don’t demand...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-02-17",
"cost": {
"input": 0.3,
"output": 0.5,
"cache_read": 0.075,
"cache_write": 0.5
},
"limit": {
"context": 131072,
"output": 8192
},
"knowledge": "2024-11"
}
},
{
"id": "x-ai/grok-4",
"name": "Grok 4",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-07-09 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 64000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.75,
"cache_write_input_per_million": 15
}
}
},
"metadata": {
"description": "Grok 4 is xAI's latest reasoning model with a 256k context window. It supports parallel tool calling, structured outputs, and both image and text inputs. Note that reasoning is not...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"image",
"text",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-09",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.75,
"cache_write": 15
},
"limit": {
"context": 256000,
"output": 64000
},
"knowledge": "2025-07"
}
},
{
"id": "x-ai/grok-4-fast",
"name": "Grok 4 Fast",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-08-19 00:00:00 UTC",
"context_window": 2000000,
"max_output_tokens": 30000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.5,
"cache_read_input_per_million": 0.05,
"cache_write_input_per_million": 0.05
}
}
},
"metadata": {
"description": "Grok 4 Fast is xAI's latest multimodal model with SOTA cost-efficiency and a 2M token context window. It comes in two flavors: non-reasoning and reasoning. Read more about the model...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 2000000,
"max_completion_tokens": 30000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-19",
"cost": {
"input": 0.2,
"output": 0.5,
"cache_read": 0.05,
"cache_write": 0.05
},
"limit": {
"context": 2000000,
"output": 30000
},
"knowledge": "2024-11"
}
},
{
"id": "x-ai/grok-4.1-fast",
"name": "Grok 4.1 Fast",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-11-19 00:00:00 UTC",
"context_window": 2000000,
"max_output_tokens": 30000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 0.5,
"cache_read_input_per_million": 0.05,
"cache_write_input_per_million": 0.05
}
}
},
"metadata": {
"description": "Grok 4.1 Fast is xAI's best agentic tool calling model that shines in real-world use cases like customer support and deep research. 2M context window. Reasoning can be enabled/disabled using...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 2000000,
"max_completion_tokens": 30000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-11-19",
"cost": {
"input": 0.2,
"output": 0.5,
"cache_read": 0.05,
"cache_write": 0.05
},
"limit": {
"context": 2000000,
"output": 30000
},
"knowledge": "2024-11"
}
},
{
"id": "x-ai/grok-4.20",
"name": "xAI: Grok 4.20",
"provider": "openrouter",
"family": "x-ai",
"created_at": "2026-03-31 17:43:39 UTC",
"context_window": 2000000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Grok 4.20 is xAI's newest flagship model with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering consistently...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 2000000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
]
}
},
{
"id": "x-ai/grok-4.20-beta",
"name": "Grok 4.20 Beta",
"provider": "openrouter",
"family": "grok",
"created_at": "2026-03-12 00:00:00 UTC",
"context_window": 2000000,
"max_output_tokens": 30000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-12",
"status": "beta",
"cost": {
"input": 2,
"output": 6,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 12
}
},
"limit": {
"context": 2000000,
"output": 30000
}
}
},
{
"id": "x-ai/grok-4.20-multi-agent",
"name": "xAI: Grok 4.20 Multi-Agent",
"provider": "openrouter",
"family": "x-ai",
"created_at": "2026-03-31 17:45:58 UTC",
"context_window": 2000000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 6.0,
"cache_read_input_per_million": 0.19999999999999998
}
}
},
"metadata": {
"description": "Grok 4.20 Multi-Agent is a variant of xAI’s Grok 4.20 designed for collaborative, agent-based workflows. Multiple agents operate in parallel to conduct deep research, coordinate tool use, and synthesize information...",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"text",
"image",
"file"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 2000000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"temperature",
"top_logprobs",
"top_p"
]
}
},
{
"id": "x-ai/grok-4.20-multi-agent-beta",
"name": "Grok 4.20 Multi - Agent Beta",
"provider": "openrouter",
"family": "grok",
"created_at": "2026-03-12 00:00:00 UTC",
"context_window": 2000000,
"max_output_tokens": 30000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 6,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-12",
"status": "beta",
"cost": {
"input": 2,
"output": 6,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 12
}
},
"limit": {
"context": 2000000,
"output": 30000
}
}
},
{
"id": "x-ai/grok-4.3",
"name": "Grok 4.3",
"provider": "openrouter",
"family": "grok",
"created_at": "2026-05-01 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 1000000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"description": "Grok 4.3 is a reasoning model from xAI. It accepts text and image inputs with text output, and is suited for agentic workflows, instruction-following tasks, and applications requiring high factual...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logprobs",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-05-01",
"cost": {
"input": 1.25,
"output": 2.5,
"cache_read": 0.2,
"context_over_200k": {
"input": 2.5,
"output": 5,
"cache_read": 0.4
}
},
"limit": {
"context": 1000000,
"output": 1000000
}
}
},
{
"id": "x-ai/grok-code-fast-1",
"name": "Grok Code Fast 1",
"provider": "openrouter",
"family": "grok",
"created_at": "2025-08-26 00:00:00 UTC",
"context_window": 256000,
"max_output_tokens": 10000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 1.5,
"cache_read_input_per_million": 0.02
}
}
},
"metadata": {
"description": "Grok Code Fast 1 is a speedy and economical reasoning model that excels at agentic coding. With reasoning traces visible in the response, developers can steer Grok Code for high-quality...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Grok",
"instruct_type": null
},
"top_provider": {
"context_length": 256000,
"max_completion_tokens": 10000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"logprobs",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-08-26",
"cost": {
"input": 0.2,
"output": 1.5,
"cache_read": 0.02
},
"limit": {
"context": 256000,
"output": 10000
},
"knowledge": "2025-08"
}
},
{
"id": "xiaomi/mimo-v2-flash",
"name": "Xiaomi: MiMo-V2-Flash",
"provider": "openrouter",
"family": "mimo",
"created_at": "2025-12-16 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 65536,
"knowledge_cutoff": "2024-12-01",
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.3,
"cache_read_input_per_million": 0.01
}
}
},
"metadata": {
"description": "MiMo-V2-Flash is an open-source foundation language model developed by Xiaomi. It is a Mixture-of-Experts model with 309B total parameters and 15B active parameters, adopting hybrid attention architecture. MiMo-V2-Flash supports a...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-04",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.1,
"output": 0.3,
"cache_read": 0.01
},
"limit": {
"context": 262144,
"output": 65536
},
"knowledge": "2024-12-01"
}
},
{
"id": "xiaomi/mimo-v2-omni",
"name": "Xiaomi: MiMo-V2-Omni",
"provider": "openrouter",
"family": "mimo",
"created_at": "2026-03-18 00:00:00 UTC",
"context_window": 262144,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2,
"cache_read_input_per_million": 0.08
}
}
},
"metadata": {
"description": "MiMo-V2-Omni is a frontier omni-modal model that natively processes image, video, and audio inputs within a unified architecture. It combines strong multimodal perception with agentic capability - visual grounding, multi-step...",
"architecture": {
"modality": "text+image+audio+video->text",
"input_modalities": [
"text",
"audio",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-18",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.4,
"output": 2,
"cache_read": 0.08
},
"limit": {
"context": 262144,
"output": 131072
},
"knowledge": "2024-12"
}
},
{
"id": "xiaomi/mimo-v2-pro",
"name": "Xiaomi: MiMo-V2-Pro",
"provider": "openrouter",
"family": "mimo",
"created_at": "2026-03-18 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 3,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"description": "MiMo-V2-Pro is Xiaomi's flagship foundation model, featuring over 1T total parameters and a 1M context length, deeply optimized for agentic scenarios. It is highly adaptable to general agent frameworks like...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-18",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 1,
"output": 3,
"cache_read": 0.2,
"context_over_200k": {
"input": 2,
"output": 6,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 131072
},
"knowledge": "2024-12"
}
},
{
"id": "xiaomi/mimo-v2.5",
"name": "Xiaomi: MiMo-V2.5",
"provider": "openrouter",
"family": "mimo",
"created_at": "2026-04-22 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.4,
"output_per_million": 2,
"cache_read_input_per_million": 0.08
}
}
},
"metadata": {
"description": "MiMo-V2.5 is a native omnimodal model by Xiaomi. It delivers Pro-level agentic performance at roughly half the inference cost, while surpassing MiMo-V2-Omni in multimodal perception across image and video understanding...",
"architecture": {
"modality": "text+image+audio+video->text",
"input_modalities": [
"text",
"audio",
"image",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-22",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.4,
"output": 2,
"cache_read": 0.08,
"context_over_200k": {
"input": 0.8,
"output": 4,
"cache_read": 0.16
}
},
"limit": {
"context": 1048576,
"output": 131072
},
"knowledge": "2024-12"
}
},
{
"id": "xiaomi/mimo-v2.5-pro",
"name": "Xiaomi: MiMo-V2.5-Pro",
"provider": "openrouter",
"family": "mimo",
"created_at": "2026-04-22 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"streaming",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 3,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"description": "MiMo-V2.5-Pro is Xiaomi’s flagship model, delivering strong performance in general agentic capabilities, complex software engineering, and long-horizon tasks, with top rankings on benchmarks such as ClawEval, GDPVal, and SWE-bench Pro....",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"response_format",
"stop",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2026-04-22",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 1,
"output": 3,
"cache_read": 0.2,
"context_over_200k": {
"input": 2,
"output": 6,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 131072
},
"knowledge": "2024-12"
}
},
{
"id": "z-ai/glm-4-32b",
"name": "Z.ai: GLM 4 32B ",
"provider": "openrouter",
"family": "z-ai",
"created_at": "2025-07-24 17:03:37 UTC",
"context_window": 128000,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.09999999999999999,
"output_per_million": 0.09999999999999999
}
}
},
"metadata": {
"description": "GLM 4 32B is a cost-effective foundation language model. It can efficiently perform complex tasks and has significantly enhanced capabilities in tool use, online search, and code-related intelligent tasks. It...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 128000,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"max_tokens",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "z-ai/glm-4.5",
"name": "GLM 4.5",
"provider": "openrouter",
"family": "glm",
"created_at": "2025-07-28 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 96000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.2
}
}
},
"metadata": {
"description": "GLM-4.5 is our latest flagship foundation model, purpose-built for agent-based applications. It leverages a Mixture-of-Experts (MoE) architecture and supports a context length of up to 128k tokens. GLM-4.5 delivers significantly...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 98304,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-28",
"cost": {
"input": 0.6,
"output": 2.2
},
"limit": {
"context": 128000,
"output": 96000
},
"knowledge": "2025-04"
}
},
{
"id": "z-ai/glm-4.5-air",
"name": "GLM 4.5 Air",
"provider": "openrouter",
"family": "glm-air",
"created_at": "2025-07-28 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 96000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.2,
"output_per_million": 1.1
}
}
},
"metadata": {
"description": "GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 98304,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-28",
"cost": {
"input": 0.2,
"output": 1.1
},
"limit": {
"context": 128000,
"output": 96000
},
"knowledge": "2025-04"
}
},
{
"id": "z-ai/glm-4.5-air:free",
"name": "GLM 4.5 Air (free)",
"provider": "openrouter",
"family": "glm-air",
"created_at": "2025-07-28 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 96000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning",
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"description": "GLM-4.5-Air is the lightweight variant of our latest flagship model family, also purpose-built for agent-centric applications. Like GLM-4.5, it adopts the Mixture-of-Experts (MoE) architecture but with a more compact parameter...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 96000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"temperature",
"tool_choice",
"tools",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-07-28",
"cost": {
"input": 0,
"output": 0
},
"limit": {
"context": 128000,
"output": 96000
},
"knowledge": "2025-04"
}
},
{
"id": "z-ai/glm-4.5v",
"name": "GLM 4.5V",
"provider": "openrouter",
"family": "glm",
"created_at": "2025-08-11 00:00:00 UTC",
"context_window": 64000,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 1.8
}
}
},
"metadata": {
"description": "GLM-4.5V is a vision-language foundation model for multimodal agent applications. Built on a Mixture-of-Experts (MoE) architecture with 106B parameters and 12B activated parameters, it achieves state-of-the-art results in video understanding,...",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 65536,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-11",
"cost": {
"input": 0.6,
"output": 1.8
},
"limit": {
"context": 64000,
"output": 16384
},
"knowledge": "2025-04"
}
},
{
"id": "z-ai/glm-4.6",
"name": "GLM 4.6",
"provider": "openrouter",
"family": "glm",
"created_at": "2025-09-30 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.2,
"cache_read_input_per_million": 0.11
}
}
},
"metadata": {
"description": "Compared with GLM-4.5, this generation brings several key improvements: Longer context window: The context window has been expanded from 128K to 200K tokens, enabling the model to handle more complex...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 204800,
"max_completion_tokens": 204800,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-30",
"cost": {
"input": 0.6,
"output": 2.2,
"cache_read": 0.11
},
"limit": {
"context": 200000,
"output": 128000
},
"knowledge": "2025-09"
}
},
{
"id": "z-ai/glm-4.6:exacto",
"name": "GLM 4.6 (exacto)",
"provider": "openrouter",
"family": "glm",
"created_at": "2025-09-30 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 1.9,
"cache_read_input_per_million": 0.11
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-30",
"cost": {
"input": 0.6,
"output": 1.9,
"cache_read": 0.11
},
"limit": {
"context": 200000,
"output": 128000
},
"knowledge": "2025-09"
}
},
{
"id": "z-ai/glm-4.6v",
"name": "Z.ai: GLM 4.6V",
"provider": "openrouter",
"family": "z-ai",
"created_at": "2025-12-08 15:24:22 UTC",
"context_window": 131072,
"max_output_tokens": 24000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 0.8999999999999999,
"cache_read_input_per_million": 0.049999999999999996
}
}
},
"metadata": {
"description": "GLM-4.6V is a large multimodal model designed for high-fidelity visual understanding and long-context reasoning across images, documents, and mixed media. It supports up to 128K tokens, processes complex page layouts...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 131072,
"max_completion_tokens": 24000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"max_tokens",
"presence_penalty",
"reasoning",
"repetition_penalty",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "z-ai/glm-4.7",
"name": "GLM-4.7",
"provider": "openrouter",
"family": "glm",
"created_at": "2025-12-22 00:00:00 UTC",
"context_window": 204800,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.6,
"output_per_million": 2.2,
"cache_read_input_per_million": 0.11
}
}
},
"metadata": {
"description": "GLM-4.7 is Z.ai’s latest flagship model, featuring upgrades in two key areas: enhanced programming capabilities and more stable multi-step reasoning/execution. It demonstrates significant improvements in executing complex agent tasks while...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 202752,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2025-12-22",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.6,
"output": 2.2,
"cache_read": 0.11
},
"limit": {
"context": 204800,
"output": 131072
},
"knowledge": "2025-04"
}
},
{
"id": "z-ai/glm-4.7-flash",
"name": "GLM-4.7-Flash",
"provider": "openrouter",
"family": "glm",
"created_at": "2026-01-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 65535,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.07,
"output_per_million": 0.4
}
}
},
"metadata": {
"description": "As a 30B-class SOTA model, GLM-4.7-Flash offers a new option that balances performance and efficiency. It is further optimized for agentic coding use cases, strengthening coding capabilities, long-horizon task planning,...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 202752,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-01-19",
"interleaved": {
"field": "reasoning_details"
},
"cost": {
"input": 0.07,
"output": 0.4
},
"limit": {
"context": 200000,
"output": 65535
}
}
},
{
"id": "z-ai/glm-5",
"name": "GLM-5",
"provider": "openrouter",
"family": "glm",
"created_at": "2026-02-12 00:00:00 UTC",
"context_window": 202752,
"max_output_tokens": 131000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 3.2,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"description": "GLM-5 is Z.ai’s flagship open-source foundation model engineered for complex systems design and long-horizon agent workflows. Built for expert developers, it delivers production-grade performance on large-scale programming tasks, rivaling leading...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 202752,
"max_completion_tokens": null,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-02-12",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 1,
"output": 3.2,
"cache_read": 0.2
},
"limit": {
"context": 202752,
"output": 131000
}
}
},
{
"id": "z-ai/glm-5-turbo",
"name": "GLM-5-Turbo",
"provider": "openrouter",
"family": "glm",
"created_at": "2026-03-16 00:00:00 UTC",
"context_window": 202752,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.96,
"output_per_million": 3.2,
"cache_read_input_per_million": 0.192
}
}
},
"metadata": {
"description": "GLM-5 Turbo is a new model from Z.ai designed for fast inference and strong performance in agent-driven environments such as OpenClaw scenarios. It is deeply optimized for real-world agent workflows...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 202752,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"max_tokens",
"min_p",
"presence_penalty",
"reasoning",
"repetition_penalty",
"response_format",
"seed",
"stop",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2026-03-16",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 0.96,
"output": 3.2,
"cache_read": 0.192,
"cache_write": 0
},
"limit": {
"context": 202752,
"output": 131072
}
}
},
{
"id": "z-ai/glm-5.1",
"name": "GLM-5.1",
"provider": "openrouter",
"family": "glm",
"created_at": "2026-04-07 00:00:00 UTC",
"context_window": 202752,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"streaming",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.4,
"output_per_million": 4.4,
"cache_read_input_per_million": 0.26
}
}
},
"metadata": {
"description": "GLM-5.1 delivers a major leap in coding capability, with particularly significant gains in handling long-horizon tasks. Unlike previous models built around minute-level interactions, GLM-5.1 can work independently and continuously on...",
"architecture": {
"modality": "text->text",
"input_modalities": [
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 202752,
"max_completion_tokens": 65535,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"parallel_tool_calls",
"presence_penalty",
"reasoning",
"reasoning_effort",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
],
"source": "models.dev",
"provider_id": "openrouter",
"open_weights": true,
"attachment": false,
"temperature": true,
"last_updated": "2026-04-07",
"interleaved": {
"field": "reasoning_content"
},
"cost": {
"input": 1.4,
"output": 4.4,
"cache_read": 0.26
},
"limit": {
"context": 202752,
"output": 131072
}
}
},
{
"id": "z-ai/glm-5v-turbo",
"name": "Z.ai: GLM 5V Turbo",
"provider": "openrouter",
"family": "z-ai",
"created_at": "2026-04-01 16:37:38 UTC",
"context_window": 202752,
"max_output_tokens": 131072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.2,
"output_per_million": 4.0,
"cache_read_input_per_million": 0.24
}
}
},
"metadata": {
"description": "GLM-5V-Turbo is Z.ai’s first native multimodal agent foundation model, built for vision-based coding and agent-driven tasks. It natively handles image, video, and text inputs, excels at long-horizon planning, complex coding,...",
"architecture": {
"modality": "text+image+video->text",
"input_modalities": [
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Other",
"instruct_type": null
},
"top_provider": {
"context_length": 202752,
"max_completion_tokens": 131072,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "~anthropic/claude-haiku-latest",
"name": "Anthropic Claude Haiku Latest",
"provider": "openrouter",
"family": "~anthropic",
"created_at": "2026-04-27 19:34:52 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.0,
"output_per_million": 5.0,
"cache_read_input_per_million": 0.09999999999999999
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the Anthropic Claude Haiku family.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 200000,
"max_completion_tokens": 64000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p"
]
}
},
{
"id": "~anthropic/claude-opus-latest",
"name": "Anthropic: Claude Opus Latest",
"provider": "openrouter",
"family": "~anthropic",
"created_at": "2026-04-21 18:16:01 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 25.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the Claude Opus family.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"tool_choice",
"tools",
"verbosity"
]
}
},
{
"id": "~anthropic/claude-sonnet-latest",
"name": "Anthropic Claude Sonnet Latest",
"provider": "openrouter",
"family": "~anthropic",
"created_at": "2026-04-27 19:32:48 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3.0,
"output_per_million": 15.0,
"cache_read_input_per_million": 0.3
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the Anthropic Claude Sonnet family.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 1000000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_p",
"verbosity"
]
}
},
{
"id": "~google/gemini-flash-latest",
"name": "Google Gemini Flash Latest",
"provider": "openrouter",
"family": "~google",
"created_at": "2026-04-27 19:33:18 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"file",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 3.0,
"cache_read_input_per_million": 0.049999999999999996,
"reasoning_output_per_million": 3.0
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the Google Gemini Flash family.",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"text",
"image",
"file",
"audio",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "~google/gemini-pro-latest",
"name": "Google Gemini Pro Latest",
"provider": "openrouter",
"family": "~google",
"created_at": "2026-04-27 19:34:11 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"audio",
"file",
"image",
"text",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2.0,
"output_per_million": 12.0,
"cache_read_input_per_million": 0.19999999999999998,
"reasoning_output_per_million": 12.0
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the Google Gemini Pro family.",
"architecture": {
"modality": "text+image+file+audio+video->text",
"input_modalities": [
"audio",
"file",
"image",
"text",
"video"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 1048576,
"max_completion_tokens": 65536,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_tokens",
"reasoning",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_p"
]
}
},
{
"id": "~moonshotai/kimi-latest",
"name": "MoonshotAI Kimi Latest",
"provider": "openrouter",
"family": "~moonshotai",
"created_at": "2026-04-27 19:33:48 UTC",
"context_window": 262144,
"max_output_tokens": 16384,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"predicted_outputs"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.75,
"output_per_million": 3.5,
"cache_read_input_per_million": 0.15
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the MoonshotAI Kimi family.",
"architecture": {
"modality": "text+image->text",
"input_modalities": [
"text",
"image"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 262144,
"max_completion_tokens": 16384,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"frequency_penalty",
"include_reasoning",
"logit_bias",
"logprobs",
"max_tokens",
"min_p",
"parallel_tool_calls",
"presence_penalty",
"reasoning",
"reasoning_effort",
"repetition_penalty",
"response_format",
"seed",
"stop",
"structured_outputs",
"temperature",
"tool_choice",
"tools",
"top_k",
"top_logprobs",
"top_p"
]
}
},
{
"id": "~openai/gpt-latest",
"name": "OpenAI GPT Latest",
"provider": "openrouter",
"family": "~openai",
"created_at": "2026-04-27 19:32:14 UTC",
"context_window": 1050000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"file",
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5.0,
"output_per_million": 30.0,
"cache_read_input_per_million": 0.5
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the OpenAI GPT family.",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 1050000,
"max_completion_tokens": 128000,
"is_moderated": true
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "~openai/gpt-mini-latest",
"name": "OpenAI GPT Mini Latest",
"provider": "openrouter",
"family": "~openai",
"created_at": "2026-04-27 19:34:31 UTC",
"context_window": 400000,
"max_output_tokens": 128000,
"knowledge_cutoff": null,
"modalities": {
"input": [
"file",
"image",
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.75,
"output_per_million": 4.5,
"cache_read_input_per_million": 0.075
}
}
},
"metadata": {
"description": "This model always redirects to the latest model in the OpenAI GPT Mini family.",
"architecture": {
"modality": "text+image+file->text",
"input_modalities": [
"file",
"image",
"text"
],
"output_modalities": [
"text"
],
"tokenizer": "Router",
"instruct_type": null
},
"top_provider": {
"context_length": 400000,
"max_completion_tokens": 128000,
"is_moderated": false
},
"per_request_limits": null,
"supported_parameters": [
"include_reasoning",
"max_completion_tokens",
"max_tokens",
"reasoning",
"response_format",
"seed",
"structured_outputs",
"tool_choice",
"tools"
]
}
},
{
"id": "sonar",
"name": "Sonar",
"provider": "perplexity",
"family": "sonar",
"created_at": "2024-01-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": "2025-09-01",
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "perplexity",
"open_weights": false,
"attachment": false,
"temperature": true,
"last_updated": "2025-09-01",
"cost": {
"input": 1,
"output": 1
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2025-09-01"
}
},
{
"id": "sonar-deep-research",
"name": "Perplexity Sonar Deep Research",
"provider": "perplexity",
"family": null,
"created_at": "2025-02-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 8,
"reasoning_output_per_million": 3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "perplexity",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-09-01",
"cost": {
"input": 2,
"output": 8,
"reasoning": 3
},
"limit": {
"context": 128000,
"output": 32768
},
"knowledge": "2025-01"
}
},
{
"id": "sonar-pro",
"name": "Sonar Pro",
"provider": "perplexity",
"family": "sonar-pro",
"created_at": "2024-01-01 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2025-09-01",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "perplexity",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-01",
"cost": {
"input": 3,
"output": 15
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2025-09-01"
}
},
{
"id": "sonar-reasoning",
"name": "sonar-reasoning",
"provider": "perplexity",
"family": null,
"created_at": null,
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"vision",
"reasoning"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.0,
"output_per_million": 5.0
}
}
},
"metadata": {}
},
{
"id": "sonar-reasoning-pro",
"name": "Sonar Reasoning Pro",
"provider": "perplexity",
"family": "sonar-reasoning",
"created_at": "2024-01-01 00:00:00 UTC",
"context_window": 128000,
"max_output_tokens": 4096,
"knowledge_cutoff": "2025-09-01",
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 8
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "perplexity",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-01",
"cost": {
"input": 2,
"output": 8
},
"limit": {
"context": 128000,
"output": 4096
},
"knowledge": "2025-09-01"
}
},
{
"id": "claude-3-5-haiku@20241022",
"name": "Claude Haiku 3.5",
"provider": "vertexai",
"family": "claude-haiku",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.8,
"output_per_million": 4,
"cache_read_input_per_million": 0.08,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 0.8,
"output": 4,
"cache_read": 0.08,
"cache_write": 1
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-07-31"
}
},
{
"id": "claude-3-5-sonnet@20241022",
"name": "Claude Sonnet 3.5 v2",
"provider": "vertexai",
"family": "claude-sonnet",
"created_at": "2024-10-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 8192,
"knowledge_cutoff": "2024-04-30",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 8192
},
"knowledge": "2024-04-30"
}
},
{
"id": "claude-3-7-sonnet@20250219",
"name": "Claude Sonnet 3.7",
"provider": "vertexai",
"family": "claude-sonnet",
"created_at": "2025-02-19 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2024-10-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-02-19",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2024-10-31"
}
},
{
"id": "claude-haiku-4-5@20251001",
"name": "Claude Haiku 4.5",
"provider": "vertexai",
"family": "claude-haiku",
"created_at": "2025-10-15 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-02-28",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1,
"output_per_million": 5,
"cache_read_input_per_million": 0.1,
"cache_write_input_per_million": 1.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-10-15",
"cost": {
"input": 1,
"output": 5,
"cache_read": 0.1,
"cache_write": 1.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-02-28"
}
},
{
"id": "claude-opus-4-1@20250805",
"name": "Claude Opus 4.1",
"provider": "vertexai",
"family": "claude-opus",
"created_at": "2025-08-05 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-08-05",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-5@20251101",
"name": "Claude Opus 4.5",
"provider": "vertexai",
"family": "claude-opus",
"created_at": "2025-11-01 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-01",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-opus-4-6@default",
"name": "Claude Opus 4.6",
"provider": "vertexai",
"family": "claude-opus",
"created_at": "2026-02-05 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2025-05-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25,
"context_over_200k": {
"input": 10,
"output": 37.5,
"cache_read": 1,
"cache_write": 12.5
}
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2025-05-31"
}
},
{
"id": "claude-opus-4-7@default",
"name": "Claude Opus 4.7",
"provider": "vertexai",
"family": "claude-opus",
"created_at": "2026-04-16 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 128000,
"knowledge_cutoff": "2026-01-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 5,
"output_per_million": 25,
"cache_read_input_per_million": 0.5,
"cache_write_input_per_million": 6.25
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": false,
"last_updated": "2026-04-16",
"cost": {
"input": 5,
"output": 25,
"cache_read": 0.5,
"cache_write": 6.25,
"context_over_200k": {
"input": 10,
"output": 37.5,
"cache_read": 1,
"cache_write": 12.5
}
},
"limit": {
"context": 1000000,
"output": 128000
},
"knowledge": "2026-01-31"
}
},
{
"id": "claude-opus-4@20250514",
"name": "Claude Opus 4",
"provider": "vertexai",
"family": "claude-opus",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 32000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 15,
"output_per_million": 75,
"cache_read_input_per_million": 1.5,
"cache_write_input_per_million": 18.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 15,
"output": 75,
"cache_read": 1.5,
"cache_write": 18.75
},
"limit": {
"context": 200000,
"output": 32000
},
"knowledge": "2025-03-31"
}
},
{
"id": "claude-sonnet-4-5@20250929",
"name": "Claude Sonnet 4.5",
"provider": "vertexai",
"family": "claude-sonnet",
"created_at": "2025-09-29 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-07-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-29",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-07-31"
}
},
{
"id": "claude-sonnet-4-6@default",
"name": "Claude Sonnet 4.6",
"provider": "vertexai",
"family": "claude-sonnet",
"created_at": "2026-02-17 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-08-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-13",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75,
"context_over_200k": {
"input": 6,
"output": 22.5,
"cache_read": 0.6,
"cache_write": 7.5
}
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-08-31"
}
},
{
"id": "claude-sonnet-4@20250514",
"name": "Claude Sonnet 4",
"provider": "vertexai",
"family": "claude-sonnet",
"created_at": "2025-05-22 00:00:00 UTC",
"context_window": 200000,
"max_output_tokens": 64000,
"knowledge_cutoff": "2025-03-31",
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 3,
"output_per_million": 15,
"cache_read_input_per_million": 0.3,
"cache_write_input_per_million": 3.75
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-22",
"cost": {
"input": 3,
"output": 15,
"cache_read": 0.3,
"cache_write": 3.75
},
"limit": {
"context": 200000,
"output": 64000
},
"knowledge": "2025-03-31"
}
},
{
"id": "gemini-1.5-flash",
"name": "Gemini 1.5 Flash",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2024-05-14 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3,
"cache_read_input_per_million": 0.01875
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-05-14",
"cost": {
"input": 0.075,
"output": 0.3,
"cache_read": 0.01875
},
"limit": {
"context": 1000000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "gemini-1.5-flash-002",
"name": "gemini-1.5-flash-002",
"provider": "vertexai",
"family": "gemini-1.5",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "gemini-1.5-flash-8b",
"name": "Gemini 1.5 Flash-8B",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2024-10-03 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.0375,
"output_per_million": 0.15,
"cache_read_input_per_million": 0.01
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-10-03",
"cost": {
"input": 0.0375,
"output": 0.15,
"cache_read": 0.01
},
"limit": {
"context": 1000000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "gemini-1.5-pro",
"name": "Gemini 1.5 Pro",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2024-02-15 00:00:00 UTC",
"context_window": 1000000,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 5,
"cache_read_input_per_million": 0.3125
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-02-15",
"cost": {
"input": 1.25,
"output": 5,
"cache_read": 0.3125
},
"limit": {
"context": 1000000,
"output": 8192
},
"knowledge": "2024-04"
}
},
{
"id": "gemini-1.5-pro-002",
"name": "gemini-1.5-pro-002",
"provider": "vertexai",
"family": "gemini-1.5",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": "PROPRIETARY",
"launch_stage": "GA",
"supported_actions": {
"openNotebook": {
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/use-cases/retail/product_attributes_extraction.ipynb"
}
},
"title": "Open Notebook",
"resourceTitle": "Notebook",
"resourceUseCase": "Product Attributes Extraction",
"resourceDescription": "Extract product descriptions and attribute json from images using Gemini 1.5 Pro. This notebook also shows the use of self-correcting prompt to improve the quality of the output."
},
"openGenerationAiStudio": {
"references": {
"us-central1": {
"uri": "https://console.cloud.google.com/vertex-ai/studio/freeform?model=gemini-1.5-pro-002"
}
}
},
"openEvaluationPipeline": {
"references": {
"us-central1": {
"uri": "https://console.cloud.google.com/vertex-ai/pipelines/vertex-ai-templates/autosxs-template"
}
},
"title": "Evaluate"
},
"openNotebooks": {
"notebooks": [
{
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_1_5_pro.ipynb"
}
},
"title": "Open Notebook"
},
{
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_1_5_pro.ipynb"
}
},
"title": "Open Notebook",
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex AI Gemini API 1.5 Pro",
"resourceDescription": "Use the Vertex AI Gemini API 1.5 Pro model to process images, video, audio, and text simultaneously."
},
{
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/use-cases/retail/product_attributes_extraction.ipynb"
}
},
"title": "Open Notebook",
"resourceTitle": "Notebook",
"resourceUseCase": "Product Attributes Extraction",
"resourceDescription": "Extract product descriptions and attribute json from images using Gemini 1.5 Pro. This notebook also shows the use of self-correcting prompt to improve the quality of the output."
}
]
}
},
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-1.5-pro-002@default"
}
},
{
"id": "gemini-2.0-flash",
"name": "Gemini 2.0 Flash",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2024-12-11 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": {
"openNotebook": {
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_2_0_flash.ipynb"
}
},
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex Serving",
"resourceDescription": "Intro to Gemini 2.0 Flash."
},
"openGenerationAiStudio": {
"references": {
"us-central1": {
"uri": "https://console.cloud.google.com/vertex-ai/generative/multimodal/create/text?model=gemini-2.0-flash-001"
}
}
},
"openNotebooks": {
"notebooks": [
{
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_2_0_flash.ipynb"
}
},
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex Serving",
"resourceDescription": "Intro to Gemini 2.0 Flash."
}
]
}
},
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.0-flash@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-11",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 8192
},
"knowledge": "2024-06"
}
},
{
"id": "gemini-2.0-flash-001",
"name": "gemini-2.0-flash-001",
"provider": "vertexai",
"family": "gemini-2",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": {
"openNotebook": {
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_2_0_flash.ipynb"
}
},
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex Serving",
"resourceDescription": "Intro to Gemini 2.0 Flash."
},
"openGenerationAiStudio": {
"references": {
"us-central1": {
"uri": "https://console.cloud.google.com/vertex-ai/generative/multimodal/create/text?model=gemini-2.0-flash-001"
}
}
},
"openNotebooks": {
"notebooks": [
{
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_2_0_flash.ipynb"
}
},
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex Serving",
"resourceDescription": "Intro to Gemini 2.0 Flash."
}
]
}
},
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.0-flash-001@default"
}
},
{
"id": "gemini-2.0-flash-exp",
"name": "gemini-2.0-flash-exp",
"provider": "vertexai",
"family": "gemini-2",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "gemini-2.0-flash-lite",
"name": "Gemini 2.0 Flash Lite",
"provider": "vertexai",
"family": "gemini-flash-lite",
"created_at": "2024-12-11 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 8192,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.075,
"output_per_million": 0.3
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2024-12-11",
"cost": {
"input": 0.075,
"output": 0.3
},
"limit": {
"context": 1048576,
"output": 8192
},
"knowledge": "2024-06"
}
},
{
"id": "gemini-2.0-flash-lite-001",
"name": "gemini-2.0-flash-lite-001",
"provider": "vertexai",
"family": "gemini-2",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.0-flash-lite-001@default"
}
},
{
"id": "gemini-2.5-flash",
"name": "Gemini 2.5 Flash",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.075,
"cache_write_input_per_million": 0.383
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-flash@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.075,
"cache_write": 0.383
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-lite",
"name": "Gemini 2.5 Flash Lite",
"provider": "vertexai",
"family": "gemini-flash-lite",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-flash-lite@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-lite-preview-06-17",
"name": "Gemini 2.5 Flash Lite Preview 06-17",
"provider": "vertexai",
"family": "gemini-flash-lite",
"created_at": "2025-06-17 00:00:00 UTC",
"context_window": 65536,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-17",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 65536,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-lite-preview-09-2025",
"name": "Gemini 2.5 Flash Lite Preview 09-25",
"provider": "vertexai",
"family": "gemini-flash-lite",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-preview-04-17",
"name": "Gemini 2.5 Flash Preview 04-17",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2025-04-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.0375
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "PUBLIC_PREVIEW",
"supported_actions": {
"openGenerationAiStudio": {
"references": {
"us-central1": {
"uri": "https://console.cloud.google.com/vertex-ai/generative/multimodal/create/text?model=gemini-2.5-flash-preview-04-17"
}
}
}
},
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-flash-preview-04-17@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-04-17",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.0375
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-preview-05-20",
"name": "Gemini 2.5 Flash Preview 05-20",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15,
"output_per_million": 0.6,
"cache_read_input_per_million": 0.0375
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-20",
"cost": {
"input": 0.15,
"output": 0.6,
"cache_read": 0.0375
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-preview-09-2025",
"name": "Gemini 2.5 Flash Preview 09-25",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.075,
"cache_write_input_per_million": 0.383
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.075,
"cache_write": 0.383
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-flash-tts",
"name": "gemini-2.5-flash-tts",
"provider": "vertexai",
"family": "gemini-2",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-flash-tts@default"
}
},
{
"id": "gemini-2.5-pro",
"name": "Gemini 2.5 Pro",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2025-03-20 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.125
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-pro@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.125,
"context_over_200k": {
"input": 2.5,
"output": 15,
"cache_read": 0.25
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro-exp-03-25",
"name": "gemini-2.5-pro-exp-03-25",
"provider": "vertexai",
"family": "gemini-2",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "EXPERIMENTAL",
"supported_actions": {
"openNotebook": {
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_2_5_pro.ipynb"
}
},
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex Serving",
"resourceDescription": "Intro to Gemini 2.5 Pro."
},
"openGenerationAiStudio": {
"references": {
"us-central1": {
"uri": "https://console.cloud.google.com/vertex-ai/generative/multimodal/create/text?model=gemini-2.5-pro-exp-03-25"
}
}
},
"openNotebooks": {
"notebooks": [
{
"references": {
"us-central1": {
"uri": "https://colab.research.google.com/github/GoogleCloudPlatform/generative-ai/blob/main/gemini/getting-started/intro_gemini_2_5_pro.ipynb"
}
},
"resourceTitle": "Notebook",
"resourceUseCase": "Vertex Serving",
"resourceDescription": "Intro to Gemini 2.5 Pro."
}
]
}
},
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-pro-exp-03-25@default"
}
},
{
"id": "gemini-2.5-pro-preview-05-06",
"name": "Gemini 2.5 Pro Preview 05-06",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2025-05-06 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.31
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-05-06",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.31
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro-preview-06-05",
"name": "Gemini 2.5 Pro Preview 06-05",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2025-06-05 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 1.25,
"output_per_million": 10,
"cache_read_input_per_million": 0.31
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-06-05",
"cost": {
"input": 1.25,
"output": 10,
"cache_read": 0.31
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-2.5-pro-tts",
"name": "gemini-2.5-pro-tts",
"provider": "vertexai",
"family": "gemini-2",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-2.5-pro-tts@default"
}
},
{
"id": "gemini-3-flash-preview",
"name": "Gemini 3 Flash Preview",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2025-12-17 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.5,
"output_per_million": 3,
"cache_read_input_per_million": 0.05
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "PUBLIC_PREVIEW",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-3-flash-preview@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-12-17",
"cost": {
"input": 0.5,
"output": 3,
"cache_read": 0.05,
"context_over_200k": {
"input": 0.5,
"output": 3,
"cache_read": 0.05
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3-pro-preview",
"name": "Gemini 3 Pro Preview",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2025-11-18 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-11-18",
"cost": {
"input": 2,
"output": 12,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-flash-image-preview",
"name": "Gemini 3.1 Flash Image (Preview)",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2026-02-26 00:00:00 UTC",
"context_window": 131072,
"max_output_tokens": 32768,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"pdf"
],
"output": [
"text",
"image"
]
},
"capabilities": [
"reasoning",
"vision",
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 60
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "PUBLIC_PREVIEW",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-3.1-flash-image-preview@default",
"source": "models.dev",
"provider_id": "google",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-26",
"cost": {
"input": 0.25,
"output": 60
},
"limit": {
"context": 131072,
"output": 32768
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-flash-lite-preview",
"name": "Gemini 3.1 Flash Lite Preview",
"provider": "vertexai",
"family": "gemini-flash-lite",
"created_at": "2026-03-03 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.25,
"output_per_million": 1.5,
"cache_read_input_per_million": 0.025,
"cache_write_input_per_million": 1
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "PUBLIC_PREVIEW",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-3.1-flash-lite-preview@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-03-03",
"cost": {
"input": 0.25,
"output": 1.5,
"cache_read": 0.025,
"cache_write": 1
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-pro-preview",
"name": "Gemini 3.1 Pro Preview",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2026-02-19 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision",
"streaming"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "PUBLIC_PREVIEW",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-3.1-pro-preview@default",
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-19",
"cost": {
"input": 2,
"output": 12,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-3.1-pro-preview-customtools",
"name": "Gemini 3.1 Pro Preview Custom Tools",
"provider": "vertexai",
"family": "gemini-pro",
"created_at": "2026-02-19 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"video",
"audio",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 2,
"output_per_million": 12,
"cache_read_input_per_million": 0.2
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2026-02-19",
"cost": {
"input": 2,
"output": 12,
"cache_read": 0.2,
"context_over_200k": {
"input": 4,
"output": 18,
"cache_read": 0.4
}
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-embedding-001",
"name": "Gemini Embedding 001",
"provider": "vertexai",
"family": "gemini",
"created_at": "2025-05-20 00:00:00 UTC",
"context_window": 2048,
"max_output_tokens": 3072,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"embeddings"
]
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.15
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": false,
"temperature": false,
"last_updated": "2025-05-20",
"cost": {
"input": 0.15,
"output": 0
},
"limit": {
"context": 2048,
"output": 3072
},
"knowledge": "2025-05"
}
},
{
"id": "gemini-embedding-2",
"name": "gemini-embedding-2",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-embedding-2@default"
}
},
{
"id": "gemini-exp-1121",
"name": "gemini-exp-1121",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "gemini-exp-1206",
"name": "gemini-exp-1206",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "gemini-flash-latest",
"name": "Gemini Flash Latest",
"provider": "vertexai",
"family": "gemini-flash",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.3,
"output_per_million": 2.5,
"cache_read_input_per_million": 0.075,
"cache_write_input_per_million": 0.383
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.3,
"output": 2.5,
"cache_read": 0.075,
"cache_write": 0.383
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-flash-lite-latest",
"name": "Gemini Flash-Lite Latest",
"provider": "vertexai",
"family": "gemini-flash-lite",
"created_at": "2025-09-25 00:00:00 UTC",
"context_window": 1048576,
"max_output_tokens": 65536,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image",
"audio",
"video",
"pdf"
],
"output": [
"text"
]
},
"capabilities": [
"function_calling",
"reasoning",
"vision"
],
"pricing": {
"text_tokens": {
"standard": {
"input_per_million": 0.1,
"output_per_million": 0.4,
"cache_read_input_per_million": 0.025
}
}
},
"metadata": {
"source": "models.dev",
"provider_id": "google-vertex",
"open_weights": false,
"attachment": true,
"temperature": true,
"last_updated": "2025-09-25",
"cost": {
"input": 0.1,
"output": 0.4,
"cache_read": 0.025
},
"limit": {
"context": 1048576,
"output": 65536
},
"knowledge": "2025-01"
}
},
{
"id": "gemini-live-2.5-flash-native-audio",
"name": "gemini-live-2.5-flash-native-audio",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"version_id": "default",
"open_source_category": null,
"launch_stage": "GA",
"supported_actions": null,
"publisher_model_template": "projects/{project}/locations/{location}/publishers/google/models/gemini-live-2.5-flash-native-audio@default"
}
},
{
"id": "gemini-pro",
"name": "gemini-pro",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "gemini-pro-vision",
"name": "gemini-pro-vision",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "text-embedding-004",
"name": "text-embedding-004",
"provider": "vertexai",
"family": "text-embedding",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "text-embedding-005",
"name": "text-embedding-005",
"provider": "vertexai",
"family": "text-embedding",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "text-multilingual-embedding-002",
"name": "text-multilingual-embedding-002",
"provider": "vertexai",
"family": "gemini",
"created_at": null,
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [],
"output": []
},
"capabilities": [
"streaming",
"function_calling"
],
"pricing": {},
"metadata": {
"source": "known_models"
}
},
{
"id": "grok-3",
"name": "Grok 3",
"provider": "xai",
"family": "grok",
"created_at": "2025-04-04 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-3-mini",
"name": "Grok 3 Mini",
"provider": "xai",
"family": "grok",
"created_at": "2025-04-04 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4-0709",
"name": "Grok 4 0709",
"provider": "xai",
"family": "grok",
"created_at": "2025-07-09 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4-1-fast-non-reasoning",
"name": "Grok 4 1 Fast Non Reasoning",
"provider": "xai",
"family": "grok",
"created_at": "2025-11-19 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4-1-fast-reasoning",
"name": "Grok 4 1 Fast Reasoning",
"provider": "xai",
"family": "grok",
"created_at": "2025-11-19 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4-fast-non-reasoning",
"name": "Grok 4 Fast Non Reasoning",
"provider": "xai",
"family": "grok",
"created_at": "2025-09-04 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4-fast-reasoning",
"name": "Grok 4 Fast Reasoning",
"provider": "xai",
"family": "grok",
"created_at": "2025-09-04 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning",
"vision"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4.20-0309-non-reasoning",
"name": "Grok 4.20 0309 Non Reasoning",
"provider": "xai",
"family": "grok",
"created_at": "2026-03-09 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4.20-0309-reasoning",
"name": "Grok 4.20 0309 Reasoning",
"provider": "xai",
"family": "grok",
"created_at": "2026-03-09 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4.20-multi-agent-0309",
"name": "Grok 4.20 Multi Agent 0309",
"provider": "xai",
"family": "grok",
"created_at": "2026-03-09 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-4.3",
"name": "Grok 4.3",
"provider": "xai",
"family": "grok",
"created_at": "2026-04-17 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-code-fast-1",
"name": "Grok Code Fast 1",
"provider": "xai",
"family": "grok",
"created_at": "2025-08-24 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output",
"reasoning"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-imagine-image",
"name": "Grok Imagine Image",
"provider": "xai",
"family": "grok",
"created_at": "2026-01-28 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-imagine-image-pro",
"name": "Grok Imagine Image Pro",
"provider": "xai",
"family": "grok",
"created_at": "2026-01-28 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-imagine-image-quality",
"name": "Grok Imagine Image Quality",
"provider": "xai",
"family": "grok",
"created_at": "2026-04-03 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
},
{
"id": "grok-imagine-video",
"name": "Grok Imagine Video",
"provider": "xai",
"family": "grok",
"created_at": "2026-01-28 00:00:00 UTC",
"context_window": null,
"max_output_tokens": null,
"knowledge_cutoff": null,
"modalities": {
"input": [
"text"
],
"output": [
"text"
]
},
"capabilities": [
"streaming",
"function_calling",
"structured_output"
],
"pricing": {},
"metadata": {
"object": "model",
"owned_by": "xai"
}
}
]