{
  "schema_version": 1,
  "family": "lfm2_audio",
  "display_name": "LFM2.5-Audio",
  "description": "Liquid AI's LFM2.5-Audio, loaded from the llama.cpp-format GGUF components Liquid publishes. ASR: a FastConformer encoder feeds an LFM2 hybrid short-conv/attention backbone. TTS: the backbone generates audio frames through a depthformer, and an LFM2 detokenizer and ISTFT turn them into 24 kHz audio. S2S: a spoken question goes in as for ASR, and the reply comes out as interleaved text and audio.",
  "category": "asr",
  "status": "experimental",
  "tasks": [
    "asr",
    "tts",
    "s2s"
  ],
  "modes": [
    "offline",
    "streaming"
  ],
  "languages": [
    "en",
    "ja"
  ],
  "capabilities": {
    "asr": [
      "vad_chunking"
    ],
    "tts": [
      "built_in_voices",
      "long_form"
    ]
  },
  "options": {
    "request": [
      {
        "name": "language",
        "type": "string",
        "description": "Must match the checkpoint's language (en or ja) when given; each published checkpoint transcribes and speaks one language.",
        "required": false,
        "default": "auto"
      },
      {
        "name": "max_tokens",
        "type": "int",
        "description": "ASR: transcript tokens per audio chunk, 512 by default. TTS: audio frames (80 ms each) per text chunk, 512 by default. S2S: text tokens and audio frames of the reply together, 1024 by default. A transcript, a text chunk's speech or a reply that reaches it is cut off there, with a warning on stderr.",
        "required": false,
        "min": 1
      },
      {
        "name": "audio_chunk_mode",
        "type": "enum",
        "description": "Audio chunking mode. auto transcribes input up to audio_chunk_seconds whole and splits longer input at pauses found by the bundled VAD (fixed chunks when the VAD model is absent); vad always splits at pauses and skips silence; fixed cuts every audio_chunk_seconds; none transcribes the whole input at once.",
        "values": [
          "auto",
          "fixed",
          "vad",
          "none"
        ],
        "required": false,
        "default": "auto"
      },
      {
        "name": "audio_chunk_seconds",
        "type": "float",
        "description": "Longest chunk in seconds, at least 1. With fixed chunks, a final piece shorter than 1 second joins the previous chunk.",
        "required": false,
        "min": 1,
        "default": 30
      },
      {
        "name": "temperature",
        "type": "float",
        "description": "TTS and S2S: audio code sampling temperature, 0.8 by default for TTS and 1.0 for S2S; 0 picks the most likely code.",
        "required": false,
        "min": 0
      },
      {
        "name": "top_k",
        "type": "int",
        "description": "TTS and S2S: sample audio codes from the k most likely, 64 by default for TTS and 4 for S2S; 0 keeps all, 1 is greedy.",
        "required": false,
        "min": 0
      },
      {
        "name": "seed",
        "type": "int",
        "description": "TTS and S2S: sampling seed; omitted requests choose a random seed. TTS text chunk i uses seed + i.",
        "required": false,
        "min": 0
      },
      {
        "name": "text_chunk_mode",
        "type": "enum",
        "description": "TTS: long-form text chunking mode; japanese for the Japanese checkpoint and default otherwise when omitted.",
        "values": [
          "default",
          "tag_aware",
          "japanese",
          "endline"
        ],
        "required": false
      },
      {
        "name": "text_chunk_size",
        "type": "int",
        "description": "TTS: maximum Unicode codepoints per text chunk, each spoken as its own turn.",
        "required": false,
        "min": 32,
        "default": 200
      },
      {
        "name": "stream_frames_per_event",
        "type": "int",
        "description": "TTS and S2S streaming: audio frames (80 ms each) per event; each event's audio is final. ASR runs offline only.",
        "required": false,
        "min": 1,
        "default": 1
      }
    ],
    "session": [
      {
        "name": "model_gguf",
        "type": "string",
        "description": "Backbone GGUF, relative to the model directory. Defaults to the directory's only backbone GGUF.",
        "required": false,
        "default": ""
      },
      {
        "name": "mmproj_gguf",
        "type": "string",
        "description": "Audio encoder (mmproj) GGUF, relative to the model directory. Defaults to mmproj-<backbone GGUF name>, or else the directory's only mmproj GGUF.",
        "required": false,
        "default": ""
      },
      {
        "name": "vocoder_gguf",
        "type": "string",
        "description": "TTS: vocoder GGUF (depthformer), relative to the model directory. Defaults to vocoder-<backbone GGUF name>, or else the directory's only vocoder GGUF.",
        "required": false,
        "default": ""
      },
      {
        "name": "detokenizer_gguf",
        "type": "string",
        "description": "TTS: audio detokenizer GGUF, relative to the model directory. Defaults to tokenizer-<backbone GGUF name>, or else the directory's only tokenizer GGUF.",
        "required": false,
        "default": ""
      },
      {
        "name": "vad_model_path",
        "type": "string",
        "description": "Path to the Silero VAD model directory used by automatic audio chunking.",
        "required": false,
        "default": "assets/framework/models/silero_vad"
      }
    ],
    "load": []
  },
  "runtime": {
    "tags": [
      "gguf",
      "cpu",
      "metal",
      "cuda"
    ]
  },
  "packages": [
    {
      "id": "lfm2_audio_1_5b_q8_0",
      "display_name": "LFM2.5-Audio-1.5B Q8_0 GGUF",
      "default": true,
      "format": "gguf",
      "precision": "q8_0",
      "target_directory": "LFM2.5-Audio-1.5B-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-Q8_0.gguf",
        "mmproj-LFM2.5-Audio-1.5B-Q8_0.gguf",
        "vocoder-LFM2.5-Audio-1.5B-Q8_0.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-Q8_0.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-GGUF",
        "revision": "7d525f883a077e20afb782f2ff618edcae0e39e4",
        "gated": false
      }
    },
    {
      "id": "lfm2_audio_1_5b_f16",
      "display_name": "LFM2.5-Audio-1.5B F16 GGUF",
      "format": "gguf",
      "precision": "f16",
      "target_directory": "LFM2.5-Audio-1.5B-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-F16.gguf",
        "mmproj-LFM2.5-Audio-1.5B-F16.gguf",
        "vocoder-LFM2.5-Audio-1.5B-F16.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-F16.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-GGUF",
        "revision": "7d525f883a077e20afb782f2ff618edcae0e39e4",
        "gated": false
      }
    },
    {
      "id": "lfm2_audio_1_5b_q4_0",
      "display_name": "LFM2.5-Audio-1.5B Q4_0 GGUF",
      "format": "gguf",
      "precision": "q4_0",
      "target_directory": "LFM2.5-Audio-1.5B-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-Q4_0.gguf",
        "mmproj-LFM2.5-Audio-1.5B-Q4_0.gguf",
        "vocoder-LFM2.5-Audio-1.5B-Q4_0.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-Q4_0.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-GGUF",
        "revision": "7d525f883a077e20afb782f2ff618edcae0e39e4",
        "gated": false
      }
    },
    {
      "id": "lfm2_audio_1_5b_jp_q8_0",
      "display_name": "LFM2.5-Audio-1.5B-JP Q8_0 GGUF",
      "format": "gguf",
      "precision": "q8_0",
      "target_directory": "LFM2.5-Audio-1.5B-JP-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-JP-Q8_0.gguf",
        "mmproj-LFM2.5-Audio-1.5B-JP-Q8_0.gguf",
        "vocoder-LFM2.5-Audio-1.5B-JP-Q8_0.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-JP-Q8_0.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-JP-GGUF",
        "revision": "64b96718b341dbd5650f9e85627cecdcbd4ac61b",
        "gated": false
      }
    },
    {
      "id": "lfm2_audio_1_5b_jp_f16",
      "display_name": "LFM2.5-Audio-1.5B-JP F16 GGUF",
      "format": "gguf",
      "precision": "f16",
      "target_directory": "LFM2.5-Audio-1.5B-JP-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-JP-F16.gguf",
        "mmproj-LFM2.5-Audio-1.5B-JP-F16.gguf",
        "vocoder-LFM2.5-Audio-1.5B-JP-F16.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-JP-F16.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-JP-GGUF",
        "revision": "64b96718b341dbd5650f9e85627cecdcbd4ac61b",
        "gated": false
      }
    },
    {
      "id": "lfm2_audio_1_5b_jp_f32",
      "display_name": "LFM2.5-Audio-1.5B-JP F32 GGUF",
      "format": "gguf",
      "precision": "f32",
      "target_directory": "LFM2.5-Audio-1.5B-JP-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-JP-F32.gguf",
        "mmproj-LFM2.5-Audio-1.5B-JP-F32.gguf",
        "vocoder-LFM2.5-Audio-1.5B-JP-F32.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-JP-F32.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-JP-GGUF",
        "revision": "64b96718b341dbd5650f9e85627cecdcbd4ac61b",
        "gated": false
      }
    },
    {
      "id": "lfm2_audio_1_5b_jp_q4_0",
      "display_name": "LFM2.5-Audio-1.5B-JP Q4_0 GGUF",
      "format": "gguf",
      "precision": "q4_0",
      "target_directory": "LFM2.5-Audio-1.5B-JP-GGUF",
      "files": [
        "LFM2.5-Audio-1.5B-JP-Q4_0.gguf",
        "mmproj-LFM2.5-Audio-1.5B-JP-Q4_0.gguf",
        "vocoder-LFM2.5-Audio-1.5B-JP-Q4_0.gguf",
        "tokenizer-LFM2.5-Audio-1.5B-JP-Q4_0.gguf"
      ],
      "download": {
        "kind": "huggingface_snapshot",
        "repo": "LiquidAI/LFM2.5-Audio-1.5B-JP-GGUF",
        "revision": "64b96718b341dbd5650f9e85627cecdcbd4ac61b",
        "gated": false
      }
    }
  ],
  "dependencies": [],
  "ui": {
    "tags": [
      "ASR",
      "TTS",
      "GGUF",
      "Stream"
    ],
    "recommended_package": "lfm2_audio_1_5b_q8_0",
    "docs": [
      "docs/community_models/lfm2_audio.md"
    ],
    "summary": "LFM2.5-Audio transcription, speech and spoken chat (English and Japanese checkpoints) from Liquid's published GGUFs.",
    "builtin_voices": [
      "us_male",
      "us_female",
      "uk_male",
      "uk_female"
    ],
    "default_voice": "us_male"
  },
  "sources": [
    {
      "format": "gguf",
      "roots": {
        "model": "."
      },
      "files": {},
      "optional_files": {},
      "tensors": {}
    }
  ]
}
