mlboydaisuke's picture
Audio8-TTS-Preview-0.6b → Core AI: dualar int8 + codec decoder fp16 + codec encoder fp16w32, tokenizer, card
72e1c93 verified
Raw History Blame Contribute Delete
5.96 kB
{
"metadata_version": "0.2",
"kind": "tts",
"name": "Audio8-TTS-Preview-0.6b-CoreAI",
"source": {
"model_definition": "torch",
"hf_model_id": "Edge0/Audio8-TTS-Preview-0.6b",
"hf_revision": "f07040f3d151f1ba0253bfb92cb2f5dd38b44594",
"model_safetensors_sha256": "62dcff0adf6c2535b3260467a7c1d482b556da57266c96a444518b76e140d2c3",
"codec_pth_sha256": "c310505aa11fe2f6cc63b8d3130dc7e77e73227774f5c62575769b1f47a8d048",
"license": "apache-2.0"
},
"audio": {
"sample_rate": 44100,
"frame_samples": 2048,
"frames_per_second": 21.533203125
},
"graphs": {
"dualar": {
"asset": "audio8_dualar_int8_cl2048_w32.aimodel",
"functions": {
"prefill": {
"inputs": {
"codes": "[1, 11, 32] int32 (row 0 ids, rows 1-10 codebooks; pad rows: 151643 / 0)",
"pos": "[1] int32 first position"
},
"outputs": {
"logits": "[32, 4097] float16 (semantic 0..4095 then eos)",
"hidden": "[32, 896] float16"
}
},
"frame": {
"inputs": {
"codes": "[1, 11, 1] int32 (the previous frame: semantic id, 10 codebooks)",
"pos": "[1] int32 (its position)",
"noise_slow": "[2, 4097] float32 uniform draws (normal branch, RAS-high branch)",
"window": "[10] int32 RAS window (-1 = none)",
"noise_fast": "[9, 4096] float32 uniform draws",
"forced": "[11] int32 (teacher forcing)",
"use_forced": "[1] float32 0/1"
},
"outputs": {
"semantic": "[1] int32 token id (151645 = eos)",
"codes": "[10] int32 codebooks 0..9",
"logits": "[4097] float16",
"hidden": "[896] float16",
"fast_logits": "[9, 4096] float16",
"sampled_semantic": "[1] int32",
"sampled_codes": "[10] int32"
}
},
"first_frame": {
"inputs": {
"logits": "[4097] float16 (the prefill's last row)",
"hidden": "[896] float16",
"noise_slow": "…",
"window": "…",
"noise_fast": "…",
"forced": "…",
"use_forced": "…"
},
"outputs": "as frame"
}
},
"state": {
"k_cache": "[24, 1, 2, 2048, 64] float16",
"v_cache": "[24, 1, 2, 2048, 64] float16 (prefill and frame)"
},
"compression": "slow AR: int8 weight-only, symmetric with clipping, per-block-32 (input axis) on the 24 layers' linears; embeddings, codebook embeddings, the 4,097-row head, norms, the fast AR (4 layers) float16"
},
"codec_decoder": {
"asset": "audio8_codec_decoder_fp16_t160.aimodel",
"functions": {
"main": {
"inputs": {
"codes": "[1, 10, 160] int32 (right-pad with 0; codebook 0 < 4096, codebooks 1-9 < 1024, clamped in-graph)"
},
"outputs": {
"wav": "[1, 327680] float16, 44.1 kHz"
}
}
},
"note": "every op is causal: a frame's samples depend on at most the 127 frames before it (the codec transformer's window) — decode a stream in 160-frame windows and keep the last 32",
"compression": null
},
"codec_encoder": {
"asset": "audio8_codec_encoder_fp16_t216.aimodel",
"functions": {
"main": {
"inputs": {
"audio": "[1, 1, 442368] float32 mono 44.1 kHz, right-pad with 0 (216 frames = 10.03 s)"
},
"outputs": {
"codes": "[1, 10, 216] int32; keep the first ceil(samples / 2048) frames"
}
}
},
"note": "voice registration (zero-shot cloning): codes + the exact transcript make an Audio8Voice",
"compression": null
}
},
"sampling": {
"top_k": 50,
"top_p": 0.9,
"temperature": 0.7,
"max_new_tokens": 512,
"ras": {
"window": 10,
"top_p": 0.9,
"temperature": 1.0
},
"rule": "argmax(softmax(processed) / -log(u)) with u ~ U(0,1); processed = top-k/top-p mask then / temperature; a semantic id repeated within the 10-frame window is replaced by the (0.9, 1.0) draw",
"eos": 151645,
"semantic_begin": 151678,
"semantic_end": 155773,
"pad": 151643
},
"prompt": {
"spec": "conversion/audio8_tts/prompt.py (segments encoded one at a time with the tokenizer, no special tokens added)",
"tokenizer": {
"files": "tokenizer/",
"class_retag": "PreTrainedTokenizerFast -> Qwen2Tokenizer for swift-transformers"
}
},
"languages": [
"yue",
"zh",
"nl",
"en",
"fr",
"de",
"it",
"ja",
"ko",
"pl",
"es"
],
"host": {
"swift": "CoreAIKit Audio8TTS (Sources/CoreAIKit/Audio8TTS)",
"python_gate": "conversion/audio8_tts/gate_audio8.py"
},
"files": {
"audio8_dualar_int8_cl2048_w32.aimodel/main.hash": {
"sha256": "73ba75f6998692b3c361a27939f2a2db36c3735236eed96f9ea955be6c70c42a",
"bytes": 32
},
"audio8_dualar_int8_cl2048_w32.aimodel/main.mlirb": {
"sha256": "6b302bf769aaee3dbe8ae78f61a36bd95fee9183dd6fd2a44fb04b5928993620",
"bytes": 876260029
},
"audio8_dualar_int8_cl2048_w32.aimodel/metadata.json": {
"sha256": "11dd0938dfd88fea0a7f862f27297a43f31d7886ce1754c3d57b552ee8100387",
"bytes": 133
},
"audio8_codec_decoder_fp16_t160.aimodel/main.hash": {
"sha256": "41cd54a438da85d253f6a733fbb806605a38d54558c8d51ada7babe251a6dccd",
"bytes": 32
},
"audio8_codec_decoder_fp16_t160.aimodel/main.mlirb": {
"sha256": "9d23fad5cf3155adfd38d76c7e777a8012dea73ad2bc47ef069f1dc67ad3b169",
"bytes": 261101172
},
"audio8_codec_decoder_fp16_t160.aimodel/metadata.json": {
"sha256": "0221caad005eed1da346fda823fe97cfbce4cbcd8a262b3dc982279d15505f4a",
"bytes": 133
},
"audio8_codec_encoder_fp16_t216.aimodel/main.hash": {
"sha256": "c24c6679b723b2eaccede267c5bbb88a77cf6b53230321339364a0d1db9c3241",
"bytes": 32
},
"audio8_codec_encoder_fp16_t216.aimodel/main.mlirb": {
"sha256": "9c77708867a6ac00aec8ab5b8ed5999f926f4f081b2a5413d29fce81cda97b48",
"bytes": 416030239
},
"audio8_codec_encoder_fp16_t216.aimodel/metadata.json": {
"sha256": "393db855a49f603faf5e09edd9334b47ddc621a8b1a78464a966d07e1714f681",
"bytes": 133
}
},
"compilation": {
"date": "2026-09-27T23:09:59.812117+00:00",
"targets": [
"macos",
"ios (jit)"
]
}
}