{
  "schema_version": 1,
  "family": "kitten_tts2",
  "display_name": "Kitten TTS 2",
  "description": "Native multilingual Kitten TTS 2 Qwen3 speech generation, S3 meanflow decoder, and reference-audio voice cloning.",
  "category": "tts",
  "status": "experimental",
  "tasks": [
    "tts",
    "clone"
  ],
  "modes": [
    "offline"
  ],
  "languages": [
    "en", "ar", "zh", "fr", "de", "hi", "it", "pt", "ru", "es"
  ],
  "runtime": {
    "tags": [
      "gguf",
      "cpu"
    ]
  },
  "capabilities": {
    "tts": [
      "built_in_voices",
      "long_form"
    ],
    "clone": [
      "speaker_reference"
    ]
  },
  "options": {
    "request": [
      {
        "name": "voice_id",
        "type": "enum",
        "values": [
          "Bella",
          "Jasper",
          "Luna",
          "Bruno",
          "Rosie",
          "Hugo",
          "Kiki",
          "Leo",
          "Matthew",
          "Elliot",
          "Willow",
          "Dolores",
          "Victor",
          "Dante",
          "Alfred",
          "Saoirse",
          "Claire",
          "Raven",
          "Marcus",
          "Herbert",
          "Diana",
          "Laurence",
          "Maeve",
          "Walter",
          "Edith",
          "Miles",
          "Grace",
          "Reginald",
          "Iris",
          "Frank",
          "Serena",
          "Julian",
          "Eleanor",
          "Otis",
          "Vincent",
          "Martha",
          "Sable",
          "Victoria",
          "PreparedBruno",
          "Arabic", "Chinese", "French", "German", "Hindi",
          "Italian", "Portuguese", "Russian", "Spanish"
        ],
        "required": false,
        "default": "Bruno",
        "description": "Prepared voice name; select a language-named preset for non-English speech. A supplied reference clip takes precedence."
      },
      {
        "name": "preset",
        "type": "enum",
        "values": [
          "stable",
          "expressive"
        ],
        "required": false,
        "default": "stable",
        "description": "Upstream sampling preset."
      },
      {
        "name": "temperature",
        "type": "float",
        "min": 0,
        "required": false,
        "description": "Sampling temperature; zero selects greedy decoding. Otherwise defaults to the preset."
      },
      {
        "name": "top_k",
        "type": "int",
        "min": 0,
        "required": false,
        "description": "Top-k sampling cutoff; defaults to the preset."
      },
      {
        "name": "top_p",
        "type": "float",
        "min": 1e-06,
        "max": 1,
        "required": false,
        "description": "Nucleus sampling cutoff; defaults to the preset."
      },
      {
        "name": "min_p",
        "type": "float",
        "min": 0,
        "max": 1,
        "required": false,
        "description": "Minimum relative probability; defaults to the preset."
      },
      {
        "name": "repetition_penalty",
        "type": "float",
        "min": 1,
        "required": false,
        "default": 1.1,
        "description": "Penalty over the last 50 generated tokens; silence and ending tokens are exempt."
      },
      {
        "name": "max_tokens",
        "type": "int",
        "min": 1,
        "max": 8192,
        "required": false,
        "description": "Speech-token budget per text chunk. Omitted budgets scale with text length, with a minimum of 200."
      },
      {
        "name": "seed",
        "type": "int",
        "min": 0,
        "required": false,
        "description": "Seed for sampling and waveform decoding."
      },
      {
        "name": "text_chunk_size",
        "type": "int",
        "min": 32,
        "max": 1000,
        "required": false,
        "default": 380,
        "description": "Maximum Unicode codepoints per framework text chunk."
      },
      {
        "name": "text_chunk_mode",
        "type": "enum",
        "values": [
          "default",
          "tag_aware",
          "japanese",
          "endline"
        ],
        "required": false,
        "default": "tag_aware",
        "description": "Framework long-form text chunking mode."
      },
      {
        "name": "reference_text",
        "type": "string",
        "required": false,
        "description": "Transcript of the reference audio; required when cloning from a 1\u201330 second clip."
      }
    ],
    "load": [],
    "session": [
      {
        "name": "weight_type",
        "type": "enum",
        "values": [
          "native",
          "f32",
          "f16",
          "bf16",
          "q4_0",
          "q8_0"
        ],
        "required": false,
        "default": "q8_0",
        "description": "Qwen projection storage. q4_0 packs original ternary weights exactly; other sources may require q8_0. Embeddings and the tied speech head remain F16."
      }
    ]
  },
  "package_defaults": {
    "download": {
      "kind": "huggingface_snapshot",
      "repo": "dignome/kitten_tts2",
      "revision": "main",
      "gated": false
    }
  },
  "packages": [
    {
      "id": "kitten_tts2_q8_0",
      "display_name": "Kitten TTS 2 Multilingual Q8_0 GGUF",
      "default": true,
      "format": "gguf",
      "precision": "q8_0",
      "target_directory": "kitten-tts2",
      "files": [
        "kitten-tts2-native-q8-multilingual.gguf",
        "LICENSE",
        "NOTICE",
        "native/LICENSE",
        "speaker/LICENSE",
        "SHA256SUMS"
      ]
    }
  ],
  "dependencies": [],
  "ui": {
    "recommended_package": "kitten_tts2_q8_0",
    "min_vram_gb": 7,
    "tags": [
      "TTS",
      "GGUF"
    ],
    "builtin_voices": [
      "Bella",
      "Jasper",
      "Luna",
      "Bruno",
      "Rosie",
      "Hugo",
      "Kiki",
      "Leo",
      "Matthew",
      "Elliot",
      "Willow",
      "Dolores",
      "Victor",
      "Dante",
      "Alfred",
      "Saoirse",
      "Claire",
      "Raven",
      "Marcus",
      "Herbert",
      "Diana",
      "Laurence",
      "Maeve",
      "Walter",
      "Edith",
      "Miles",
      "Grace",
      "Reginald",
      "Iris",
      "Frank",
      "Serena",
      "Julian",
      "Eleanor",
      "Otis",
      "Vincent",
      "Martha",
      "Sable",
      "Victoria",
      "Arabic", "Chinese", "French", "German", "Hindi",
      "Italian", "Portuguese", "Russian", "Spanish"
    ],
    "default_voice": "Bruno",
    "docs": [
      "docs/community_models/kitten_tts2.md"
    ]
  },
  "sources": [
    {
      "format": "gguf",
      "roots": {
        "model": ".",
        "weights": "$gguf"
      },
      "files": {
        "config": "model:config.json",
        "lm_config": "model:lm/config.json",
        "tokenizer_config": "model:lm/tokenizer_config.json",
        "tokenizer_json": "model:lm/tokenizer.json",
        "default_voices": "model:cpp/default/voices.json"
      },
      "tensors": {
        "language_model": {
          "source": "weights:",
          "prefix": "language_model"
        },
        "s3gen": {
          "source": "weights:",
          "prefix": "s3gen"
        },
        "speaker": {
          "source": "weights:",
          "prefix": "speaker"
        }
      }
    },
    {
      "format": "safetensors",
      "roots": {
        "model": "."
      },
      "files": {
        "config": "model:config.json",
        "lm_config": "model:lm/config.json",
        "tokenizer_config": "model:lm/tokenizer_config.json",
        "tokenizer_json": "model:lm/tokenizer.json",
        "default_voices": "model:cpp/default/voices.json"
      },
      "tensors": {
        "language_model": "model:lm/model.safetensors",
        "s3gen": "model:native/s3gen_meanflow.safetensors",
        "speaker": "model:speaker/model.safetensors"
      }
    }
  ]
}
