{ "metadata_version": "0.2", "kind": "tts", "name": "Audio8-TTS-Preview-0.6b-CoreAI", "source": { "model_definition": "torch", "hf_model_id": "Edge0/Audio8-TTS-Preview-0.6b", "hf_revision": "f07040f3d151f1ba0253bfb92cb2f5dd38b44594", "model_safetensors_sha256": "62dcff0adf6c2535b3260467a7c1d482b556da57266c96a444518b76e140d2c3", "codec_pth_sha256": "c310505aa11fe2f6cc63b8d3130dc7e77e73227774f5c62575769b1f47a8d048", "license": "apache-2.0" }, "audio": { "sample_rate": 44100, "frame_samples": 2048, "frames_per_second": 21.533203125 }, "graphs": { "dualar": { "asset": "audio8_dualar_int8_cl2048_w32.aimodel", "functions": { "prefill": { "inputs": { "codes": "[1, 11, 32] int32 (row 0 ids, rows 1-10 codebooks; pad rows: 151643 / 0)", "pos": "[1] int32 first position" }, "outputs": { "logits": "[32, 4097] float16 (semantic 0..4095 then eos)", "hidden": "[32, 896] float16" } }, "frame": { "inputs": { "codes": "[1, 11, 1] int32 (the previous frame: semantic id, 10 codebooks)", "pos": "[1] int32 (its position)", "noise_slow": "[2, 4097] float32 uniform draws (normal branch, RAS-high branch)", "window": "[10] int32 RAS window (-1 = none)", "noise_fast": "[9, 4096] float32 uniform draws", "forced": "[11] int32 (teacher forcing)", "use_forced": "[1] float32 0/1" }, "outputs": { "semantic": "[1] int32 token id (151645 = eos)", "codes": "[10] int32 codebooks 0..9", "logits": "[4097] float16", "hidden": "[896] float16", "fast_logits": "[9, 4096] float16", "sampled_semantic": "[1] int32", "sampled_codes": "[10] int32" } }, "first_frame": { "inputs": { "logits": "[4097] float16 (the prefill's last row)", "hidden": "[896] float16", "noise_slow": "…", "window": "…", "noise_fast": "…", "forced": "…", "use_forced": "…" }, "outputs": "as frame" } }, "state": { "k_cache": "[24, 1, 2, 2048, 64] float16", "v_cache": "[24, 1, 2, 2048, 64] float16 (prefill and frame)" }, "compression": "slow AR: int8 weight-only, symmetric with clipping, per-block-32 (input axis) on the 24 layers' linears; embeddings, codebook embeddings, the 4,097-row head, norms, the fast AR (4 layers) float16" }, "codec_decoder": { "asset": "audio8_codec_decoder_fp16_t160.aimodel", "functions": { "main": { "inputs": { "codes": "[1, 10, 160] int32 (right-pad with 0; codebook 0 < 4096, codebooks 1-9 < 1024, clamped in-graph)" }, "outputs": { "wav": "[1, 327680] float16, 44.1 kHz" } } }, "note": "every op is causal: a frame's samples depend on at most the 127 frames before it (the codec transformer's window) — decode a stream in 160-frame windows and keep the last 32", "compression": null }, "codec_encoder": { "asset": "audio8_codec_encoder_fp16_t216.aimodel", "functions": { "main": { "inputs": { "audio": "[1, 1, 442368] float32 mono 44.1 kHz, right-pad with 0 (216 frames = 10.03 s)" }, "outputs": { "codes": "[1, 10, 216] int32; keep the first ceil(samples / 2048) frames" } } }, "note": "voice registration (zero-shot cloning): codes + the exact transcript make an Audio8Voice", "compression": null } }, "sampling": { "top_k": 50, "top_p": 0.9, "temperature": 0.7, "max_new_tokens": 512, "ras": { "window": 10, "top_p": 0.9, "temperature": 1.0 }, "rule": "argmax(softmax(processed) / -log(u)) with u ~ U(0,1); processed = top-k/top-p mask then / temperature; a semantic id repeated within the 10-frame window is replaced by the (0.9, 1.0) draw", "eos": 151645, "semantic_begin": 151678, "semantic_end": 155773, "pad": 151643 }, "prompt": { "spec": "conversion/audio8_tts/prompt.py (segments encoded one at a time with the tokenizer, no special tokens added)", "tokenizer": { "files": "tokenizer/", "class_retag": "PreTrainedTokenizerFast -> Qwen2Tokenizer for swift-transformers" } }, "languages": [ "yue", "zh", "nl", "en", "fr", "de", "it", "ja", "ko", "pl", "es" ], "host": { "swift": "CoreAIKit Audio8TTS (Sources/CoreAIKit/Audio8TTS)", "python_gate": "conversion/audio8_tts/gate_audio8.py" }, "files": { "audio8_dualar_int8_cl2048_w32.aimodel/main.hash": { "sha256": "73ba75f6998692b3c361a27939f2a2db36c3735236eed96f9ea955be6c70c42a", "bytes": 32 }, "audio8_dualar_int8_cl2048_w32.aimodel/main.mlirb": { "sha256": "6b302bf769aaee3dbe8ae78f61a36bd95fee9183dd6fd2a44fb04b5928993620", "bytes": 876260029 }, "audio8_dualar_int8_cl2048_w32.aimodel/metadata.json": { "sha256": "11dd0938dfd88fea0a7f862f27297a43f31d7886ce1754c3d57b552ee8100387", "bytes": 133 }, "audio8_codec_decoder_fp16_t160.aimodel/main.hash": { "sha256": "41cd54a438da85d253f6a733fbb806605a38d54558c8d51ada7babe251a6dccd", "bytes": 32 }, "audio8_codec_decoder_fp16_t160.aimodel/main.mlirb": { "sha256": "9d23fad5cf3155adfd38d76c7e777a8012dea73ad2bc47ef069f1dc67ad3b169", "bytes": 261101172 }, "audio8_codec_decoder_fp16_t160.aimodel/metadata.json": { "sha256": "0221caad005eed1da346fda823fe97cfbce4cbcd8a262b3dc982279d15505f4a", "bytes": 133 }, "audio8_codec_encoder_fp16_t216.aimodel/main.hash": { "sha256": "c24c6679b723b2eaccede267c5bbb88a77cf6b53230321339364a0d1db9c3241", "bytes": 32 }, "audio8_codec_encoder_fp16_t216.aimodel/main.mlirb": { "sha256": "9c77708867a6ac00aec8ab5b8ed5999f926f4f081b2a5413d29fce81cda97b48", "bytes": 416030239 }, "audio8_codec_encoder_fp16_t216.aimodel/metadata.json": { "sha256": "393db855a49f603faf5e09edd9334b47ddc621a8b1a78464a966d07e1714f681", "bytes": 133 } }, "compilation": { "date": "2026-09-27T23:09:59.812117+00:00", "targets": [ "macos", "ios (jit)" ] } }