stack-open-weights.stack.json json Copy{
"osr": "1",
"kind": "stack",
"name": "Open-weights models",
"description": "Hosted open-weights models only - the same stack runs on your own hardware later.",
"targets": [
{
"ocm": "1",
"id": "model-z-ai-glm-5-3",
"kind": "llm",
"name": "GLM 5.3",
"description": "GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...",
"publisher": "Z.ai",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"z-ai",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"models"
],
"languages": [
"en"
],
"modalities": [
"text"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 1310720
},
"quality_prior": 0.449,
"examples": [
"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1787086655",
"url": "https://openrouter.ai/z-ai/glm-5.3",
"key": "z-ai/glm-5.3",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "z-ai/glm-5.3",
"pricing": {
"input_usd_per_1m": 0.91,
"output_usd_per_1m": 2.86,
"cache_read_usd_per_1m": 0.169
},
"max_output_tokens": 131072,
"output_modalities": [
"text"
],
"hugging_face_id": "zai-org/GLM-5.3",
"reasoning": true,
"created": 1787086655,
"benchmarks": {
"intelligence_index": 44.9,
"coding_index": 74.8,
"agentic_index": 53.4,
"hub_downloads": 888143,
"hub_downloads_total": 888143,
"hub_trending": 48,
"hub_likes": 1867,
"osr_index": 45.9,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.001885
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-moonshotai-kimi-k3",
"kind": "llm",
"name": "Kimi K3",
"description": "Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...",
"publisher": "Moonshot AI",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"moonshotai",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"image",
"video",
"models"
],
"languages": [
"en"
],
"modalities": [
"text",
"image",
"video"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 1048576
},
"quality_prior": 0.438,
"examples": [
"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1784215858",
"url": "https://openrouter.ai/moonshotai/kimi-k3",
"key": "moonshotai/kimi-k3",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "moonshotai/kimi-k3",
"pricing": {
"input_usd_per_1m": 2.1,
"output_usd_per_1m": 10.95,
"cache_read_usd_per_1m": 0.23
},
"max_output_tokens": 943718,
"output_modalities": [
"text"
],
"hugging_face_id": "moonshotai/Kimi-K3",
"reasoning": true,
"created": 1784215858,
"benchmarks": {
"intelligence_index": 43.8,
"coding_index": 76.2,
"agentic_index": 50.6,
"hub_downloads": 2197706,
"hub_downloads_total": 4538945,
"hub_likes": 11417,
"hub_trending": 95,
"osr_index": 45.3,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.006525
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-z-ai-glm-5-3-flash",
"kind": "llm",
"name": "GLM 5.3 Flash",
"description": "GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...",
"publisher": "Z.ai",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"z-ai",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"image",
"video",
"models"
],
"languages": [
"en"
],
"modalities": [
"text",
"image",
"video"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 1310720
},
"quality_prior": 0.419,
"examples": [
"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1787752741",
"url": "https://openrouter.ai/z-ai/glm-5.3-flash",
"key": "z-ai/glm-5.3-flash",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "z-ai/glm-5.3-flash",
"pricing": {
"input_usd_per_1m": 0.09,
"output_usd_per_1m": 0.3,
"cache_read_usd_per_1m": 0.018
},
"max_output_tokens": 131072,
"output_modalities": [
"text"
],
"hugging_face_id": "zai-org/GLM-5.3-Flash",
"reasoning": true,
"created": 1787752741,
"benchmarks": {
"intelligence_index": 41.9,
"coding_index": 71.5,
"agentic_index": 51.2,
"hub_downloads": 2669173,
"hub_likes": 2441,
"hub_trending": 158,
"hub_downloads_total": 2669173,
"osr_index": 44.2,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.000195
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-qwen-qwen3-8-2-4t-a95b",
"kind": "llm",
"name": "Qwen3.8 2.4T A95B",
"description": "Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is...",
"publisher": "Qwen",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"qwen",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"models"
],
"languages": [
"en"
],
"modalities": [
"text"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 1048576
},
"quality_prior": 0.4,
"examples": [
"Qwen3.8 2.4T A95B is an open-weight sparse mixture-of-experts model from Qwen and the open-weight variant of [Qwen3.8 Max](/qwen/qwen3.8-max), with 95 billion active parameters out of 2.4 trillion total. It is..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1786551702",
"url": "https://openrouter.ai/qwen/qwen3.8-2.4t-a95b",
"key": "qwen/qwen3.8-2.4t-a95b",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "qwen/qwen3.8-2.4t-a95b",
"pricing": {
"input_usd_per_1m": 2,
"output_usd_per_1m": 6,
"cache_read_usd_per_1m": 0.25
},
"max_output_tokens": 131072,
"output_modalities": [
"text"
],
"hugging_face_id": "Qwen/Qwen3.8-2.4T-A95B",
"reasoning": true,
"created": 1786551702,
"benchmarks": {
"intelligence_index": 40,
"coding_index": 71.9,
"agentic_index": 50.4,
"hub_downloads": 56314,
"hub_downloads_total": 70796,
"hub_likes": 1232,
"hub_trending": 15,
"osr_index": 43.1,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.004
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
}
],
"rules": [
{
"name": "hard-to-strongest",
"when": {
"min_complexity": 0.7
},
"prefer": [
"model-z-ai-glm-5-3"
]
}
],
"objective": {
"quality": 0.7,
"cost": 0.8,
"latency": 0.2
},
"metadata": {
"source": {
"provider": "catalogue-stacks",
"path": "open-weights",
"ref": "2026-09-19",
"url": "https://openrouter.ai/models",
"key": "catalogue-stacks/open-weights",
"catalogue": "https://openrouter.ai/api/v1",
"generated": true
}
}
}