| { |
| "name": "cosyvoice3-coreml", |
| "version": "1.0.0", |
| "language": "zh", |
| "library": "fluidaudio", |
| "description": "CoreML conversions of CosyVoice3 Mandarin TTS (Qwen2-0.5B LLM + Flow mel generator + HiFT vocoder).", |
| "pipeline_tag": "text-to-speech", |
| "sample_rate_hz": 24000, |
| "compute": { |
| "target_platform": "Apple Silicon (M-series)", |
| "min_os": "macOS 14 / iOS 17", |
| "neural_engine": ["LLM-Prefill", "LLM-Decode", "HiFT"], |
| "cpu_and_gpu": ["Flow"] |
| }, |
| "model_graph": { |
| "llm_hidden_dim": 896, |
| "llm_layers": 24, |
| "llm_query_heads": 14, |
| "llm_kv_heads": 2, |
| "llm_head_dim": 64, |
| "llm_text_vocab": 151936, |
| "speech_vocab": 6761, |
| "speech_sos": 6561, |
| "speech_eos": 6562, |
| "speech_task_id": 6563, |
| "mel_bins": 80, |
| "mel_hop": 480, |
| "mel_nfft": 1920 |
| }, |
| "models": [ |
| { |
| "name": "LLM-Prefill-T256-M768-fp16", |
| "paths": { |
| "mlpackage": "LLM-Prefill-T256-M768-fp16.mlpackage", |
| "mlmodelc": "LLM-Prefill-T256-M768-fp16.mlmodelc" |
| }, |
| "dtype": "fp16", |
| "compute_units": "cpuAndNeuralEngine", |
| "purpose": "Qwen2 prefill over 256-token context, initializes 768-slot KV cache.", |
| "size_bytes": 729042944, |
| "inputs": { |
| "inputs_embeds": "[1, 256, 896] fp16", |
| "attention_mask": "[1, 256] int32", |
| "position_ids": "[1, 256] int32" |
| }, |
| "outputs": { |
| "logits": "[1, 256, 6761] fp16 (speech vocab)", |
| "kv_k_out": "[24, 1, 2, 768, 64] fp16", |
| "kv_v_out": "[24, 1, 2, 768, 64] fp16" |
| } |
| }, |
| { |
| "name": "LLM-Decode-M768-fp16", |
| "paths": { |
| "mlpackage": "LLM-Decode-M768-fp16.mlpackage", |
| "mlmodelc": "LLM-Decode-M768-fp16.mlmodelc" |
| }, |
| "dtype": "fp16", |
| "compute_units": "cpuAndNeuralEngine", |
| "purpose": "Single-step AR decode against a 768-slot KV cache.", |
| "size_bytes": 728567808, |
| "inputs": { |
| "inputs_embeds": "[1, 1, 896] fp16", |
| "cur_len": "[1] int32", |
| "kv_k_in": "[24, 1, 2, 768, 64] fp16", |
| "kv_v_in": "[24, 1, 2, 768, 64] fp16" |
| }, |
| "outputs": { |
| "logits": "[1, 1, 6761] fp16", |
| "kv_k_out": "[24, 1, 2, 768, 64] fp16", |
| "kv_v_out": "[24, 1, 2, 768, 64] fp16" |
| } |
| }, |
| { |
| "name": "Flow-N250-fp16", |
| "paths": { |
| "mlpackage": "Flow-N250-fp16.mlpackage", |
| "mlmodelc": "Flow-N250-fp16.mlmodelc" |
| }, |
| "dtype": "fp16", |
| "compute_units": "cpuAndGPU", |
| "purpose": "Speech tokens -> 80-bin log-mel @ 24 kHz. Must run with cpuAndGPU: pure CPU overflows the fused LayerNorm and produces NaNs; ANE refuses to compile this graph (ANECCompile fails). GPU path uses fp32 accumulators internally and is stable + ~3x faster than the previous fp32/cpuOnly shipping config.", |
| "size_bytes": 669208054, |
| "inputs": { |
| "token_total": "[1, 250] int32 (prompt_ids || new_ids, right-padded)", |
| "num_prompt_tokens": "[1] int32", |
| "prompt_feat": "[1, 500, 80] fp32 (right-padded)", |
| "embedding": "[1, 192] fp32 (CAMPPlus speaker embedding)" |
| }, |
| "outputs": { |
| "mel": "[1, 80, 500] fp32 (full buffer; slice to num_prompt_mel..num_prompt_mel+2*N_new)", |
| "num_prompt_mel": "[1] int32" |
| } |
| }, |
| { |
| "name": "HiFT-T500-fp16", |
| "paths": { |
| "mlpackage": "HiFT-T500-fp16.mlpackage", |
| "mlmodelc": "HiFT-T500-fp16.mlmodelc" |
| }, |
| "dtype": "fp16", |
| "compute_units": "cpuAndNeuralEngine", |
| "purpose": "Mel -> 24 kHz PCM via iSTFT-based vocoder.", |
| "size_bytes": 46448640, |
| "inputs": { |
| "mel": "[1, 80, 500] fp16 (right-padded)", |
| "num_valid_frames": "[1] int32" |
| }, |
| "outputs": { |
| "audio": "[1, 240000] fp16 (clip to 480 * num_valid_frames samples)" |
| } |
| } |
| ], |
| "embeddings": [ |
| { |
| "name": "embeddings-runtime-fp32", |
| "path": "embeddings/embeddings-runtime-fp32.safetensors", |
| "shape": [151936, 896], |
| "dtype": "fp32", |
| "size_bytes": 568770400, |
| "purpose": "Qwen2 model.embed_tokens.weight at post-.float() runtime dtype. Required for bit-exact parity with Python reference. Swift mmaps this file." |
| }, |
| { |
| "name": "speech_embedding-fp16", |
| "path": "embeddings/speech_embedding-fp16.safetensors", |
| "shape": [6761, 896], |
| "dtype": "fp16", |
| "size_bytes": 12115808, |
| "purpose": "CosyVoice3 speech_embedding table. Row-lookup per decoded speech token in the decode loop." |
| } |
| ], |
| "tokenizer": { |
| "kind": "qwen2-bpe", |
| "vocab_file": "tokenizer/vocab.json", |
| "merges_file": "tokenizer/merges.txt", |
| "config_file": "tokenizer/tokenizer_config.json", |
| "special_tokens_file": "tokenizer/special_tokens.json", |
| "base_vocab_size": 151936, |
| "special_token_count": 281, |
| "special_token_id_range": [151643, 151923], |
| "required_tokens": { |
| "endofprompt": 151646, |
| "endoftext": 151643, |
| "im_start": 151644, |
| "im_end": 151645 |
| } |
| }, |
| "voices": [ |
| { |
| "voice_id": "cosyvoice3-default-zh", |
| "files": { |
| "tensors": "voices/cosyvoice3-default-zh.safetensors", |
| "metadata": "voices/cosyvoice3-default-zh.json" |
| }, |
| "reference_wav": "CosyVoice upstream zero_shot_prompt.wav", |
| "prompt_utterance": "希望你以后能够做的比我还好呦。", |
| "n_speech": 87, |
| "mel_frames": 174, |
| "size_bytes": 57244 |
| } |
| ], |
| "additional_voices_repo": "FluidInference/cosyvoice3-voices-zh", |
| "swift": { |
| "library": "FluidAudio", |
| "manager": "CosyVoice3TtsManager", |
| "public_api": "synthesize(text: String, promptAssets: CosyVoice3PromptAssets) async throws -> SynthesisResult", |
| "default_voice": "voices/cosyvoice3-default-zh.safetensors" |
| }, |
| "license": "Apache-2.0", |
| "upstream": "FunAudioLLM/CosyVoice3" |
| } |
|
|