stack-deepseek.stack.json json Copy{
"osr": "1",
"kind": "stack",
"name": "DeepSeek tiers",
"description": "DeepSeek only: DeepSeek V4 Flash 0423 for everyday requests, DeepSeek V3.1 for moderate work, DeepSeek V4.1 Flash for hard ones - with complexity rules and a cost-aware objective.",
"targets": [
{
"ocm": "1",
"id": "model-deepseek-deepseek-v4-flash",
"kind": "llm",
"name": "DeepSeek V4 Flash 0423",
"description": "DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and...",
"publisher": "DeepSeek",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"deepseek",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"models"
],
"languages": [
"en"
],
"modalities": [
"text"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 1048576
},
"quality_prior": 0.248,
"examples": [
"DeepSeek V4 Flash is an efficiency-optimized Mixture-of-Experts model from DeepSeek with 284B total parameters and 13B activated parameters, supporting a 1M-token context window. It is designed for fast inference and..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1777000666",
"url": "https://openrouter.ai/deepseek/deepseek-v4-flash",
"key": "deepseek/deepseek-v4-flash",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "deepseek/deepseek-v4-flash",
"pricing": {
"input_usd_per_1m": 0.04956,
"output_usd_per_1m": 0.09912,
"cache_read_usd_per_1m": 0.009912
},
"max_output_tokens": 384000,
"output_modalities": [
"text"
],
"hugging_face_id": "deepseek-ai/DeepSeek-V4-Flash",
"reasoning": true,
"created": 1777000666,
"benchmarks": {
"intelligence_index": 24.8,
"coding_index": 52,
"agentic_index": 27.9,
"hub_downloads": 1614751,
"hub_downloads_total": 11921343,
"hub_likes": 2236,
"hub_trending": 17,
"osr_index": 34.4,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.000074
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-deepseek-deepseek-chat-v3-1",
"kind": "llm",
"name": "DeepSeek V3.1",
"description": "DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context...",
"publisher": "DeepSeek",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"deepseek",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"models"
],
"languages": [
"en"
],
"modalities": [
"text"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 163840
},
"quality_prior": 0.6,
"examples": [
"DeepSeek-V3.1 is a large hybrid reasoning model (671B parameters, 37B active) that supports both thinking and non-thinking modes via prompt templates. It extends the DeepSeek-V3 base with a two-phase long-context..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1755779628",
"url": "https://openrouter.ai/deepseek/deepseek-chat-v3.1",
"key": "deepseek/deepseek-chat-v3.1",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "deepseek/deepseek-chat-v3.1",
"pricing": {
"input_usd_per_1m": 0.25,
"output_usd_per_1m": 0.95,
"cache_read_usd_per_1m": 0.13
},
"max_output_tokens": 32768,
"output_modalities": [
"text"
],
"hugging_face_id": "deepseek-ai/DeepSeek-V3.1",
"reasoning": true,
"created": 1755779628,
"benchmarks": {
"hub_downloads": 269115,
"hub_downloads_total": 2998988,
"hub_likes": 833,
"hub_trending": 1
}
},
"cost": {
"usd_per_1k_tokens": 0.0006
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-deepseek-deepseek-v4-1-flash",
"kind": "llm",
"name": "DeepSeek V4.1 Flash",
"description": "DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on...",
"publisher": "DeepSeek",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"deepseek",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"image",
"models"
],
"languages": [
"en"
],
"modalities": [
"text",
"image"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 1048576
},
"quality_prior": 0.395,
"examples": [
"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the first built on the company's Causal Encoder-Decoder (CED) architecture. It activates 8B parameters on input and 16B on..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1789021285",
"url": "https://openrouter.ai/deepseek/deepseek-v4.1-flash",
"key": "deepseek/deepseek-v4.1-flash",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "deepseek/deepseek-v4.1-flash",
"pricing": {
"input_usd_per_1m": 0.15,
"output_usd_per_1m": 0.6,
"cache_read_usd_per_1m": 0.003
},
"max_output_tokens": 384000,
"output_modalities": [
"text"
],
"hugging_face_id": "deepseek-ai/DeepSeek-V4.1-Flash",
"reasoning": true,
"created": 1789021285,
"benchmarks": {
"intelligence_index": 39.5,
"hub_downloads": 429865,
"hub_downloads_total": 429865,
"hub_likes": 3135,
"hub_trending": 1050,
"osr_index": 42.8,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.000375
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
}
],
"rules": [
{
"name": "hard-to-strongest",
"when": {
"min_complexity": 0.7
},
"prefer": [
"model-deepseek-deepseek-v4-1-flash"
]
},
{
"name": "easy-stays-cheap",
"when": {
"max_complexity": 0.3
},
"avoid": [
"model-deepseek-deepseek-v4-1-flash"
]
}
],
"objective": {
"quality": 0.7,
"cost": 0.8,
"latency": 0.2
},
"metadata": {
"source": {
"provider": "catalogue-stacks",
"path": "deepseek",
"ref": "2026-09-19",
"url": "https://openrouter.ai/deepseek",
"key": "catalogue-stacks/deepseek",
"catalogue": "https://openrouter.ai/api/v1",
"generated": true
}
}
}