{
  "schema_version": 1,
  "family": "nemotron_asr",
  "display_name": "Nemotron 3.5 ASR",
  "description": "NVIDIA 600M streaming ASR model for low-latency and batch transcription across 40 language-locales, with native punctuation, capitalization, automatic language detection, and configurable chunk sizes.",
  "category": "asr",
  "status": "supported",
  "tasks": [
    "asr"
  ],
  "modes": [
    "offline",
    "streaming"
  ],
  "languages": [
    "ar-AR",
    "bg-BG",
    "cs-CZ",
    "da-DK",
    "de-DE",
    "el-GR",
    "en-GB",
    "en-US",
    "es-ES",
    "es-US",
    "et-EE",
    "fi-FI",
    "fr-CA",
    "fr-FR",
    "he-IL",
    "hi-IN",
    "hr-HR",
    "hu-HU",
    "it-IT",
    "ja-JP",
    "ko-KR",
    "lt-LT",
    "lv-LV",
    "mt-MT",
    "nb-NO",
    "nl-NL",
    "nn-NO",
    "pl-PL",
    "pt-BR",
    "pt-PT",
    "ro-RO",
    "ru-RU",
    "sk-SK",
    "sl-SI",
    "sv-SE",
    "th-TH",
    "tr-TR",
    "uk-UA",
    "vi-VN",
    "zh-CN"
  ],
  "capabilities": {},
  "dependencies": [],
  "options": {
    "request": [
      {
        "name": "language",
        "type": "string",
        "description": "ASR prompt language such as en-US, da-DK, or auto; default is the model's default prompt.",
        "required": false
      },
      {
        "name": "lookahead_tokens",
        "type": "int",
        "description": "Chunk-limited encoder right context in 80 ms frames; supported values and the default come from processor_config (0, 3, 6 or 13; default 3 for the published model).",
        "required": false,
        "min": 0
      },
      {
        "name": "max_tokens",
        "type": "int",
        "description": "Maximum RNNT generated tokens; 0 or omitted uses the model-derived limit.",
        "required": false,
        "min": 0,
        "default": 0
      },
      {
        "name": "keep_language_tags",
        "type": "bool",
        "description": "Keep language tag tokens in decoded text; default false.",
        "required": false,
        "default": false
      },
      {
        "name": "return_timestamps",
        "type": "bool",
        "description": "Return token timestamps; set by the CLI for --words-out. Token timestamps are always computed.",
        "required": false,
        "default": false
      },
      {
        "name": "streaming",
        "type": "bool",
        "description": "Request a streaming run; must be false for offline sessions.",
        "required": false,
        "default": false
      },
      {
        "name": "speaker_probabilities",
        "type": "path",
        "description": "nemotron_3_diar speaker_probabilities.safetensors for the same audio (diarize with return_frame_probabilities=true). Setting it enables speaker-tagged output: speaker turns with text and a SegLST artifact.",
        "required": false
      },
      {
        "name": "speaker_mode",
        "type": "enum",
        "values": [
          "masked",
          "attribution"
        ],
        "description": "masked runs one ASR stream per active speaker on speaker-masked features (NeMo masked multitalker ASR) and transcribes overlapping speech; attribution assigns each word of one transcript to the most active speaker and is cheaper.",
        "required": false,
        "default": "masked"
      },
      {
        "name": "speaker_mask",
        "type": "enum",
        "values": [
          "mel",
          "audio"
        ],
        "description": "Masked mode only: mask log-mel features (NeMo mask_features) or zero the waveform of inactive 80 ms frames.",
        "required": false,
        "default": "mel"
      },
      {
        "name": "speaker_segment_gap_sec",
        "type": "float",
        "description": "Start a new segment for a speaker after a pause longer than this; default 1.0 (NeMo sent_break_sec).",
        "required": false,
        "min": 0.0,
        "default": 1.0
      },
      {
        "name": "speaker_segment_max_sec",
        "type": "float",
        "description": "Streaming only: force-close a speaker-turn event after this many seconds; the final result keeps the full segments. Default 10.",
        "required": false,
        "min": 0.001,
        "default": 10.0
      }
    ],
    "session": [
      {
        "name": "weight_type",
        "type": "enum",
        "description": "Shared matmul weight storage type; default native.",
        "preset": "weight_type_full",
        "required": false,
        "default": "native"
      },
      {
        "name": "matmul_weight_type",
        "type": "enum",
        "description": "Encoder and decoder matmul weight storage type (native, f32, f16, bf16 or q8_0); defaults to weight_type.",
        "preset": "weight_type_full",
        "required": false
      },
      {
        "name": "conv_weight_type",
        "type": "enum",
        "description": "Convolution weight storage type (native, f32 or f16); default native.",
        "preset": "weight_type_conv",
        "required": false,
        "default": "native"
      },
      {
        "name": "weight_context_mb",
        "type": "int",
        "description": "Weight context arena size in MiB; default 3072.",
        "required": false,
        "min": 1,
        "default": 3072
      },
      {
        "name": "encoder_graph_arena_mb",
        "type": "int",
        "description": "Encoder graph arena size in MiB; default 1024.",
        "required": false,
        "min": 1,
        "default": 1024
      },
      {
        "name": "decoder_graph_arena_mb",
        "type": "int",
        "description": "Decoder graph arena size in MiB; default 256.",
        "required": false,
        "min": 1,
        "default": 256
      },
      {
        "name": "mem_saver",
        "type": "bool",
        "description": "Release the offline encoder graph after each offline request; default false.",
        "required": false,
        "default": false
      }
    ],
    "load": []
  },
  "runtime": {
    "tags": [
      "gguf",
      "stream"
    ]
  },
  "ui": {
    "recommended_package": "nemotron_asr_q8_0",
    "tags": [
      "ASR",
      "GGUF",
      "Stream"
    ],
    "docs": [
      "docs/asr.md",
      "docs/gguf.md"
    ]
  },
  "package_defaults": {
    "download": {
      "kind": "huggingface_snapshot",
      "repo": "audio-cpp/audio.cpp-gguf",
      "revision": "main",
      "gated": false
    }
  },
  "packages": [
    {
      "id": "nemotron_asr_q8_0",
      "display_name": "Nemotron 3.5 ASR Streaming 0.6B Q8_0 GGUF",
      "default": true,
      "format": "gguf",
      "precision": "q8_0",
      "target_directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF",
      "files": [
        "Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf"
      ],
      "strip_prefix": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF"
    },
    {
      "id": "nemotron_asr_f16",
      "display_name": "Nemotron 3.5 ASR Streaming 0.6B F16 GGUF",
      "format": "gguf",
      "precision": "f16",
      "target_directory": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF",
      "files": [
        "Nemotron-3.5-ASR-Streaming-0.6B-GGUF/nemotron-3.5-asr-streaming-0.6b-f16.gguf"
      ],
      "strip_prefix": "Nemotron-3.5-ASR-Streaming-0.6B-GGUF"
    },
    {
      "id": "nemotron_asr_safetensors",
      "display_name": "Nemotron 3.5 ASR Streaming 0.6B Safetensors",
      "format": "safetensors",
      "precision": "native",
      "target_directory": "nemotron-3.5-asr-streaming-0.6b",
      "files": [
        "config.json",
        "model.safetensors",
        "processor_config.json",
        "tokenizer.json"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "nvidia/nemotron-3.5-asr-streaming-0.6b"
      }
    }
  ],
  "sources": [
    {
      "format": "gguf",
      "roots": {
        "model": ".",
        "weights": "$gguf"
      },
      "files": {
        "config": "model:config.json",
        "processor_config": "model:processor_config.json",
        "tokenizer_json": "model:tokenizer.json"
      },
      "tensors": {
        "weights": "weights:"
      }
    },
    {
      "format": "safetensors",
      "roots": {
        "model": "."
      },
      "files": {
        "config": "model:config.json",
        "processor_config": "model:processor_config.json",
        "tokenizer_json": "model:tokenizer.json"
      },
      "tensors": {
        "weights": "model:model.safetensors"
      }
    }
  ]
}
