stack-nvidia.stack.json json Copy{
"osr": "1",
"kind": "stack",
"name": "NVIDIA tiers",
"description": "NVIDIA only: Nemotron 3.5 Content Safety for everyday requests, Nemotron 3 Nano 30B A3B for moderate work, Nemotron 3 Ultra for hard ones - with complexity rules and a cost-aware objective.",
"targets": [
{
"ocm": "1",
"id": "model-nvidia-nemotron-3-5-content-safety",
"kind": "llm",
"name": "Nemotron 3.5 Content Safety",
"description": "NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...",
"publisher": "NVIDIA",
"version": "1.0.0",
"capabilities": {
"domains": [
"math"
],
"tags": [
"llm",
"nvidia",
"openrouter",
"open-weights",
"reasoning",
"image",
"models"
],
"languages": [
"en"
],
"modalities": [
"text",
"image"
],
"supports_tools": false,
"supports_streaming": true,
"context_window": 131072
},
"quality_prior": 0.6,
"examples": [
"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1780581864",
"url": "https://openrouter.ai/nvidia/nemotron-3.5-content-safety",
"key": "nvidia/nemotron-3.5-content-safety",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "nvidia/nemotron-3.5-content-safety",
"pricing": {
"input_usd_per_1m": 0.2,
"output_usd_per_1m": 0.2
},
"max_output_tokens": 117964,
"output_modalities": [
"text"
],
"hugging_face_id": "nvidia/Nemotron-3.5-Content-Safety",
"reasoning": true,
"created": 1780581864,
"benchmarks": {
"hub_downloads": 110991,
"hub_downloads_total": 312090,
"hub_likes": 51,
"hub_trending": 0
}
},
"cost": {
"usd_per_1k_tokens": 0.0002
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-nvidia-nemotron-3-nano-30b-a3b",
"kind": "llm",
"name": "Nemotron 3 Nano 30B A3B",
"description": "NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully...",
"publisher": "NVIDIA",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"nvidia",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"models"
],
"languages": [
"en"
],
"modalities": [
"text"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 262144
},
"quality_prior": 0.089,
"examples": [
"NVIDIA Nemotron 3 Nano 30B A3B is a small language MoE model with highest compute efficiency and accuracy for developers to build specialized agentic AI systems. The model is fully..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1765731275",
"url": "https://openrouter.ai/nvidia/nemotron-3-nano-30b-a3b",
"key": "nvidia/nemotron-3-nano-30b-a3b",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "nvidia/nemotron-3-nano-30b-a3b",
"pricing": {
"input_usd_per_1m": 0.06,
"output_usd_per_1m": 0.24
},
"max_output_tokens": 235929,
"output_modalities": [
"text"
],
"hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16",
"reasoning": true,
"created": 1765731275,
"benchmarks": {
"intelligence_index": 8.9,
"coding_index": 14.4,
"agentic_index": 1,
"hub_downloads": 692217,
"hub_downloads_total": 9494975,
"hub_likes": 822,
"hub_trending": 3,
"osr_index": 25.4,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.00015
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
},
{
"ocm": "1",
"id": "model-nvidia-nemotron-3-ultra-550b-a55b",
"kind": "llm",
"name": "Nemotron 3 Ultra",
"description": "NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...",
"publisher": "NVIDIA",
"version": "1.0.0",
"capabilities": {
"domains": [
"general"
],
"tags": [
"llm",
"nvidia",
"openrouter",
"open-weights",
"reasoning",
"tool-calling",
"models"
],
"languages": [
"en"
],
"modalities": [
"text"
],
"supports_tools": true,
"supports_streaming": true,
"context_window": 262144
},
"quality_prior": 0.234,
"examples": [
"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it..."
],
"primary": false,
"metadata": {
"source": {
"provider": "models",
"ref": "1780551208",
"url": "https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b",
"key": "nvidia/nemotron-3-ultra-550b-a55b",
"catalogue": "https://openrouter.ai/api/v1"
},
"model": "nvidia/nemotron-3-ultra-550b-a55b",
"pricing": {
"input_usd_per_1m": 0.625,
"output_usd_per_1m": 3.125,
"cache_read_usd_per_1m": 0.1875
},
"max_output_tokens": 32768,
"output_modalities": [
"text"
],
"hugging_face_id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
"reasoning": true,
"created": 1780551208,
"benchmarks": {
"intelligence_index": 23.4,
"coding_index": 49.3,
"agentic_index": 21.7,
"hub_downloads_total": 992107,
"hub_likes": 346,
"hub_trending": 0,
"hub_downloads": 174742,
"osr_index": 33.6,
"osr_coverage": 0.4
}
},
"cost": {
"usd_per_1k_tokens": 0.001875
},
"endpoints": [
{
"protocol": "openai",
"url": "https://openrouter.ai/api/v1"
}
]
}
],
"rules": [
{
"name": "hard-to-strongest",
"when": {
"min_complexity": 0.7
},
"prefer": [
"model-nvidia-nemotron-3-ultra-550b-a55b"
]
},
{
"name": "easy-stays-cheap",
"when": {
"max_complexity": 0.3
},
"avoid": [
"model-nvidia-nemotron-3-ultra-550b-a55b"
]
}
],
"objective": {
"quality": 0.7,
"cost": 0.8,
"latency": 0.2
},
"metadata": {
"source": {
"provider": "catalogue-stacks",
"path": "nvidia",
"ref": "2026-09-19",
"url": "https://openrouter.ai/nvidia",
"key": "catalogue-stacks/nvidia",
"catalogue": "https://openrouter.ai/api/v1",
"generated": true
}
}
}