CosyVoice3-0.5B-coreml / manifest.json
alexwengg's picture
Upload 14 files
d4a719e verified
Raw
History Blame Contribute Delete
5.86 kB
{
"name": "cosyvoice3-coreml",
"version": "1.0.0",
"language": "zh",
"library": "fluidaudio",
"description": "CoreML conversions of CosyVoice3 Mandarin TTS (Qwen2-0.5B LLM + Flow mel generator + HiFT vocoder).",
"pipeline_tag": "text-to-speech",
"sample_rate_hz": 24000,
"compute": {
"target_platform": "Apple Silicon (M-series)",
"min_os": "macOS 14 / iOS 17",
"neural_engine": ["LLM-Prefill", "LLM-Decode", "HiFT"],
"cpu_and_gpu": ["Flow"]
},
"model_graph": {
"llm_hidden_dim": 896,
"llm_layers": 24,
"llm_query_heads": 14,
"llm_kv_heads": 2,
"llm_head_dim": 64,
"llm_text_vocab": 151936,
"speech_vocab": 6761,
"speech_sos": 6561,
"speech_eos": 6562,
"speech_task_id": 6563,
"mel_bins": 80,
"mel_hop": 480,
"mel_nfft": 1920
},
"models": [
{
"name": "LLM-Prefill-T256-M768-fp16",
"paths": {
"mlpackage": "LLM-Prefill-T256-M768-fp16.mlpackage",
"mlmodelc": "LLM-Prefill-T256-M768-fp16.mlmodelc"
},
"dtype": "fp16",
"compute_units": "cpuAndNeuralEngine",
"purpose": "Qwen2 prefill over 256-token context, initializes 768-slot KV cache.",
"size_bytes": 729042944,
"inputs": {
"inputs_embeds": "[1, 256, 896] fp16",
"attention_mask": "[1, 256] int32",
"position_ids": "[1, 256] int32"
},
"outputs": {
"logits": "[1, 256, 6761] fp16 (speech vocab)",
"kv_k_out": "[24, 1, 2, 768, 64] fp16",
"kv_v_out": "[24, 1, 2, 768, 64] fp16"
}
},
{
"name": "LLM-Decode-M768-fp16",
"paths": {
"mlpackage": "LLM-Decode-M768-fp16.mlpackage",
"mlmodelc": "LLM-Decode-M768-fp16.mlmodelc"
},
"dtype": "fp16",
"compute_units": "cpuAndNeuralEngine",
"purpose": "Single-step AR decode against a 768-slot KV cache.",
"size_bytes": 728567808,
"inputs": {
"inputs_embeds": "[1, 1, 896] fp16",
"cur_len": "[1] int32",
"kv_k_in": "[24, 1, 2, 768, 64] fp16",
"kv_v_in": "[24, 1, 2, 768, 64] fp16"
},
"outputs": {
"logits": "[1, 1, 6761] fp16",
"kv_k_out": "[24, 1, 2, 768, 64] fp16",
"kv_v_out": "[24, 1, 2, 768, 64] fp16"
}
},
{
"name": "Flow-N250-fp16",
"paths": {
"mlpackage": "Flow-N250-fp16.mlpackage",
"mlmodelc": "Flow-N250-fp16.mlmodelc"
},
"dtype": "fp16",
"compute_units": "cpuAndGPU",
"purpose": "Speech tokens -> 80-bin log-mel @ 24 kHz. Must run with cpuAndGPU: pure CPU overflows the fused LayerNorm and produces NaNs; ANE refuses to compile this graph (ANECCompile fails). GPU path uses fp32 accumulators internally and is stable + ~3x faster than the previous fp32/cpuOnly shipping config.",
"size_bytes": 669208054,
"inputs": {
"token_total": "[1, 250] int32 (prompt_ids || new_ids, right-padded)",
"num_prompt_tokens": "[1] int32",
"prompt_feat": "[1, 500, 80] fp32 (right-padded)",
"embedding": "[1, 192] fp32 (CAMPPlus speaker embedding)"
},
"outputs": {
"mel": "[1, 80, 500] fp32 (full buffer; slice to num_prompt_mel..num_prompt_mel+2*N_new)",
"num_prompt_mel": "[1] int32"
}
},
{
"name": "HiFT-T500-fp16",
"paths": {
"mlpackage": "HiFT-T500-fp16.mlpackage",
"mlmodelc": "HiFT-T500-fp16.mlmodelc"
},
"dtype": "fp16",
"compute_units": "cpuAndNeuralEngine",
"purpose": "Mel -> 24 kHz PCM via iSTFT-based vocoder.",
"size_bytes": 46448640,
"inputs": {
"mel": "[1, 80, 500] fp16 (right-padded)",
"num_valid_frames": "[1] int32"
},
"outputs": {
"audio": "[1, 240000] fp16 (clip to 480 * num_valid_frames samples)"
}
}
],
"embeddings": [
{
"name": "embeddings-runtime-fp32",
"path": "embeddings/embeddings-runtime-fp32.safetensors",
"shape": [151936, 896],
"dtype": "fp32",
"size_bytes": 568770400,
"purpose": "Qwen2 model.embed_tokens.weight at post-.float() runtime dtype. Required for bit-exact parity with Python reference. Swift mmaps this file."
},
{
"name": "speech_embedding-fp16",
"path": "embeddings/speech_embedding-fp16.safetensors",
"shape": [6761, 896],
"dtype": "fp16",
"size_bytes": 12115808,
"purpose": "CosyVoice3 speech_embedding table. Row-lookup per decoded speech token in the decode loop."
}
],
"tokenizer": {
"kind": "qwen2-bpe",
"vocab_file": "tokenizer/vocab.json",
"merges_file": "tokenizer/merges.txt",
"config_file": "tokenizer/tokenizer_config.json",
"special_tokens_file": "tokenizer/special_tokens.json",
"base_vocab_size": 151936,
"special_token_count": 281,
"special_token_id_range": [151643, 151923],
"required_tokens": {
"endofprompt": 151646,
"endoftext": 151643,
"im_start": 151644,
"im_end": 151645
}
},
"voices": [
{
"voice_id": "cosyvoice3-default-zh",
"files": {
"tensors": "voices/cosyvoice3-default-zh.safetensors",
"metadata": "voices/cosyvoice3-default-zh.json"
},
"reference_wav": "CosyVoice upstream zero_shot_prompt.wav",
"prompt_utterance": "希望你以后能够做的比我还好呦。",
"n_speech": 87,
"mel_frames": 174,
"size_bytes": 57244
}
],
"additional_voices_repo": "FluidInference/cosyvoice3-voices-zh",
"swift": {
"library": "FluidAudio",
"manager": "CosyVoice3TtsManager",
"public_api": "synthesize(text: String, promptAssets: CosyVoice3PromptAssets) async throws -> SynthesisResult",
"default_voice": "voices/cosyvoice3-default-zh.safetensors"
},
"license": "Apache-2.0",
"upstream": "FunAudioLLM/CosyVoice3"
}