{
  "schema_version": 1,
  "family": "echo_tts",
  "display_name": "Echo-TTS",
  "description": "Echo-TTS is an English zero-shot voice-cloning TTS model packaged for audio.cpp. A 2.8B diffusion transformer generates 80-D latents in PCA space which the Fish S1-DAC decodes to 44.1 kHz audio. Generation is a fixed 29.72 s window (640 latents).",
  "category": "tts",
  "status": "experimental",
  "tasks": [
    "clone"
  ],
  "modes": [
    "offline"
  ],
  "languages": [
    "en"
  ],
  "runtime": {
    "tags": [
      "gguf"
    ]
  },
  "capabilities": {
    "clone": [
      "speaker_reference"
    ]
  },
  "options": {
    "request": [
      {
        "name": "text_guidance_scale",
        "type": "float",
        "description": "Classifier-free guidance scale on the text condition.",
        "required": false,
        "min": 0.0,
        "default": 3.0
      },
      {
        "name": "speaker_guidance_scale",
        "type": "float",
        "description": "Classifier-free guidance scale on the speaker condition.",
        "required": false,
        "min": 0.0,
        "default": 8.0
      },
      {
        "name": "num_inference_steps",
        "type": "int",
        "description": "Euler sampler steps.",
        "required": false,
        "min": 1,
        "default": 40
      },
      {
        "name": "truncation_factor",
        "type": "float",
        "description": "Initial-noise truncation factor.",
        "required": false,
        "min": 0.0,
        "max": 1.0,
        "default": 0.8
      },
      {
        "name": "speaker_kv_scale",
        "type": "float",
        "description": "Force-speaker KV scaling. 1.0 disables; 1.5 is the upstream default when enabled.",
        "required": false,
        "min": 1.0,
        "default": 1.0
      },
      {
        "name": "seed",
        "type": "int",
        "description": "RNG seed for the initial latent.",
        "required": false,
        "default": 0
      },
      {
        "name": "reference_duration_sec",
        "type": "float",
        "required": false,
        "description": "Trim the speaker reference to at most this many seconds before encoding. Shorter references are cheaper and often clone better; upstream's guidance favours around 10 s. Defaults to 15 s. Values above the trained maximum of 297.1 s are clamped. Set per request, or as a session default from CLI or server config.",
        "default": 15.0
      },
      {
        "name": "max_duration_sec",
        "type": "float",
        "description": "Cap the generation window, in seconds (up to 29.7215 s, the trained maximum; larger values are clamped to it). Quantised down to a whole latent frame of 46.44 ms. Left unset, the window is estimated per chunk from text length and widened automatically if the utterance does not finish inside it, which is substantially cheaper for short text.",
        "required": false,
        "min": 0.05,
        "max": 29.7215
      },
      {
        "name": "guidance_interval",
        "type": "int",
        "description": "Refresh the two unconditional CFG lanes only every Nth step inside the guidance window, reusing the previous guidance correction in between. 1 reproduces upstream exactly. Worth raising to 2 at 30+ steps, where the correction is still measured 8-10 times across the guided phase (roughly 1.3x fewer denoiser evaluations); not recommended below ~20 steps, where it is already coarsely sampled and staleness shows up as degraded timbre and text adherence.",
        "required": false,
        "min": 1,
        "default": 1
      }
    ],
    "session": [
      {
        "name": "reference_duration_sec",
        "type": "float",
        "required": false,
        "description": "Trim the speaker reference to at most this many seconds before encoding. Shorter references are cheaper and often clone better; upstream's guidance favours around 10 s. Defaults to 15 s. Values above the trained maximum of 297.1 s are clamped. Set per request, or as a session default from CLI or server config.",
        "default": 15.0
      },
      {
        "name": "reference_cache_slots",
        "type": "int",
        "required": false,
        "description": "How many encoded speaker references to keep. Encoding is linear in reference length, so reusing a voice across requests avoids repeating it. 0 disables caching; default 4."
      },
      {
        "name": "mem_saver",
        "type": "bool",
        "required": false,
        "description": "Release Fish S1-DAC decode graph space before encoding reference speaker audio and encode graph space after. Avoids having both allocations resident simulaneously at the cost of rebuilding the graphs when a speaker cache-miss occurs.",
        "default": false
      }
    ],
    "load": []
  },
  "packages": [
    {
      "id": "echo_tts_q8_0",
      "display_name": "Echo-TTS Q8_0 GGUF",
      "default": true,
      "format": "gguf",
      "precision": "q8_0",
      "target_directory": "Echo-TTS-GGUF",
      "files": [
        "echo-tts-q8_0.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "dignome/Echo-TTS",
        "revision": "919681c9906c88963352d5f96e0f732ddd3989ce",
        "gated": false
      }
    },
    {
      "id": "echo_tts_f16",
      "display_name": "Echo-TTS F16 GGUF",
      "format": "gguf",
      "precision": "f16",
      "target_directory": "Echo-TTS-GGUF",
      "files": [
        "echo-tts-f16.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "dignome/Echo-TTS",
        "revision": "919681c9906c88963352d5f96e0f732ddd3989ce",
        "gated": false
      }
    }
  ],
  "dependencies": [],
  "ui": {
    "recommended_package": "echo_tts_q8_0",
    "tags": [
      "TTS",
      "Clone",
      "GGUF"
    ],
    "docs": []
  },
  "sources": [
    {
      "format": "gguf",
      "roots": {
        "model": ".",
        "weights": "$gguf"
      },
      "tensors": {
        "dit_weights": {
          "source": "weights:",
          "prefix": "dit_weights"
        },
        "pca": {
          "source": "weights:",
          "prefix": "pca"
        },
        "codec_weights": {
          "source": "weights:",
          "prefix": "ae"
        }
      }
    }
  ]
}
