{
  "schema_version": 1,
  "family": "nemotron_3_diar",
  "display_name": "Nemotron 3 Diarization",
  "description": "NVIDIA eight-speaker streaming diarization model with arrival-order speaker identities and configurable latency profiles.",
  "category": "speech_analysis",
  "status": "supported",
  "tasks": [
    "diar"
  ],
  "modes": [
    "offline",
    "streaming"
  ],
  "languages": [
    "multilingual"
  ],
  "capabilities": {
    "diar": [
      "speaker_turns"
    ]
  },
  "dependencies": [],
  "options": {
    "request": [
      {
        "name": "speaker_threshold",
        "type": "float",
        "description": "Speaker activity threshold used for turn decoding.",
        "required": false,
        "min": 0.0,
        "max": 1.0,
        "default": 0.5
      },
      {
        "name": "speaker_min_frames",
        "type": "int",
        "description": "Minimum decoded turn duration in 10 ms output frames.",
        "required": false,
        "min": 0,
        "default": 0
      },
      {
        "name": "speaker_pad_frames",
        "type": "int",
        "description": "Padding applied to decoded turns in 10 ms output frames.",
        "required": false,
        "min": 0,
        "default": 0
      },
      {
        "name": "return_frame_probabilities",
        "type": "bool",
        "description": "Attach the raw 10 ms speaker-activity timeline as the speaker_probabilities.safetensors artifact (F32 [frames, speakers] plus streaming-geometry metadata), the input for nemotron_asr speaker tagging.",
        "required": false,
        "default": false
      }
    ],
    "session": [
      {
        "name": "latency_profile",
        "type": "enum",
        "description": "Streaming latency profile: the official very_high, low, very_low and ultra_low presets, asr_la0/3/6/13 for nemotron_asr speaker masking at that ASR lookahead, or custom.",
        "values": [
          "very_high",
          "low",
          "very_low",
          "ultra_low",
          "asr_la0",
          "asr_la3",
          "asr_la6",
          "asr_la13",
          "custom"
        ],
        "required": false,
        "default": "very_high"
      },
      {
        "name": "spkcache_len",
        "type": "int",
        "description": "Speaker-cache length in 80 ms encoder frames for the custom profile.",
        "required": false,
        "min": 16
      },
      {
        "name": "fifo_len",
        "type": "int",
        "description": "FIFO length in 80 ms encoder frames for the custom profile.",
        "required": false,
        "min": 0
      },
      {
        "name": "chunk_len",
        "type": "int",
        "description": "Chunk length in 80 ms encoder frames for the custom profile.",
        "required": false,
        "min": 1
      },
      {
        "name": "chunk_right_context",
        "type": "int",
        "description": "Future context in 80 ms encoder frames for the custom profile.",
        "required": false,
        "min": 0
      },
      {
        "name": "spkcache_update_period",
        "type": "int",
        "description": "FIFO-to-cache update period in 80 ms encoder frames for the custom profile.",
        "required": false,
        "min": 1
      },
      {
        "name": "attention",
        "type": "enum",
        "values": ["auto", "flash", "eager"],
        "description": "Encoder attention lowering; auto uses flash attention where the backend supports it.",
        "required": false,
        "default": "auto"
      },
      {
        "name": "graph_arena_mb",
        "type": "int",
        "description": "Inference graph arena size in MiB.",
        "required": false,
        "min": 1,
        "default": 1024
      },
      {
        "name": "weight_context_mb",
        "type": "int",
        "description": "Weight context size in MiB.",
        "required": false,
        "min": 1,
        "default": 1024
      },
      {
        "name": "weight_type",
        "type": "enum",
        "description": "Default model weight storage type.",
        "preset": "weight_type_full",
        "required": false,
        "default": "native"
      }
    ],
    "load": []
  },
  "runtime": {
    "tags": [
      "gguf",
      "stream"
    ]
  },
  "ui": {
    "recommended_package": "nemotron_3_diar_bf16",
    "tags": [
      "Diar",
      "Stream",
      "GGUF"
    ],
    "docs": [
      "docs/models/nemotron_3_diar.md",
      "docs/speech_analysis.md",
      "docs/gguf.md"
    ]
  },
  "package_defaults": {
    "download": {
      "kind": "huggingface_snapshot",
      "repo": "audio-cpp/Nemotron-3-Diarization-GGUF",
      "revision": "main",
      "gated": false
    }
  },
  "packages": [
    {
      "id": "nemotron_3_diar_bf16",
      "display_name": "Nemotron 3 Diarization BF16 GGUF",
      "default": true,
      "format": "gguf",
      "precision": "bf16",
      "target_directory": "Nemotron-3-Diarization-GGUF",
      "files": [
        "nemotron-3-diarization-bf16.gguf"
      ]
    },
    {
      "id": "nemotron_3_diar_q8_0",
      "display_name": "Nemotron 3 Diarization Q8_0 GGUF",
      "format": "gguf",
      "precision": "q8_0",
      "target_directory": "Nemotron-3-Diarization-GGUF",
      "files": [
        "nemotron-3-diarization-q8_0.gguf"
      ]
    }
  ],
  "sources": [
    {
      "format": "gguf",
      "roots": {
        "model": ".",
        "weights": "$gguf"
      },
      "files": {
        "config": "model:config.json",
        "processor": "model:processor_config.json"
      },
      "tensors": {
        "weights": "weights:"
      }
    },
    {
      "format": "safetensors",
      "roots": {
        "model": "."
      },
      "files": {
        "config": "model:config.json",
        "processor": "model:processor_config.json"
      },
      "tensors": {
        "weights": "model:model.safetensors"
      }
    }
  ]
}
