Present DuDE as full-duplex conversation TTS with named samples and verified held-out provenance
Browse files- .gitattributes +2 -0
- CITATION.cff +1 -1
- README.md +107 -119
- dude_tts/interleave.py +1 -1
- examples/examples.json +12 -8
- examples/{V01_S0192_I00000307-0000-student.wav → sample_1.wav} +0 -0
- examples/{V01_S0192_I00000307-0000.xml → sample_1.xml} +0 -0
- examples/{V01_S0220_I00000126-0000-student.wav → sample_2.wav} +0 -0
- examples/{V01_S0220_I00000126-0000.xml → sample_2.xml} +0 -0
- provenance.json +85 -0
- release.json +13 -10
- verification.json +8 -6
- voices/presets.json +4 -4
.gitattributes
CHANGED
|
@@ -39,3 +39,5 @@ voices/voice_1.wav filter=lfs diff=lfs merge=lfs -text
|
|
| 39 |
voices/voice_2.wav filter=lfs diff=lfs merge=lfs -text
|
| 40 |
voices/voice_3.wav filter=lfs diff=lfs merge=lfs -text
|
| 41 |
voices/voice_4.wav filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
| 39 |
voices/voice_2.wav filter=lfs diff=lfs merge=lfs -text
|
| 40 |
voices/voice_3.wav filter=lfs diff=lfs merge=lfs -text
|
| 41 |
voices/voice_4.wav filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
examples/sample_1.wav filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
examples/sample_2.wav filter=lfs diff=lfs merge=lfs -text
|
CITATION.cff
CHANGED
|
@@ -1,7 +1,7 @@
|
|
| 1 |
cff-version: 1.2.0
|
| 2 |
message: "If you use DuDE, please cite this model release."
|
| 3 |
type: software
|
| 4 |
-
title: "DuDE:
|
| 5 |
authors:
|
| 6 |
- family-names: Chang
|
| 7 |
given-names: Cheng-Kuang
|
|
|
|
| 1 |
cff-version: 1.2.0
|
| 2 |
message: "If you use DuDE, please cite this model release."
|
| 3 |
type: software
|
| 4 |
+
title: "DuDE: Full-Duplex Conversation TTS"
|
| 5 |
authors:
|
| 6 |
- family-names: Chang
|
| 7 |
given-names: Cheng-Kuang
|
README.md
CHANGED
|
@@ -4,7 +4,8 @@ language:
|
|
| 4 |
license: cc-by-nc-4.0
|
| 5 |
pipeline_tag: text-to-speech
|
| 6 |
tags:
|
| 7 |
-
- duplex
|
|
|
|
| 8 |
- dialogue
|
| 9 |
- stereo
|
| 10 |
- xml
|
|
@@ -12,34 +13,32 @@ tags:
|
|
| 12 |
base_model: Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice
|
| 13 |
datasets:
|
| 14 |
- facebook/seamless-interaction
|
| 15 |
-
model-index:
|
| 16 |
-
- name: DuDE-TTS
|
| 17 |
-
results: []
|
| 18 |
---
|
| 19 |
|
| 20 |
-
# DuDE:
|
| 21 |
|
| 22 |
-
|
| 23 |
-
|
| 24 |
-
|
|
|
|
| 25 |
|
| 26 |
-
|
| 27 |
|
| 28 |
-
|
| 29 |
-
|
| 30 |
-
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
## Featured examples and complete input XML
|
| 34 |
|
| 35 |
-
|
| 36 |
-
Both channels
|
|
|
|
| 37 |
|
| 38 |
-
###
|
| 39 |
|
| 40 |
-
<audio controls src="https://huggingface.co/penguinfish1688/duplexdataengine/resolve/main/examples/
|
| 41 |
|
| 42 |
-
[Download stereo audio](examples/
|
| 43 |
|
| 44 |
<details>
|
| 45 |
<summary>Show complete input XML</summary>
|
|
@@ -73,11 +72,11 @@ Both channels preserve their shared origin; neither lane is trimmed or shifted.
|
|
| 73 |
|
| 74 |
</details>
|
| 75 |
|
| 76 |
-
###
|
| 77 |
|
| 78 |
-
<audio controls src="https://huggingface.co/penguinfish1688/duplexdataengine/resolve/main/examples/
|
| 79 |
|
| 80 |
-
[Download stereo audio](examples/
|
| 81 |
|
| 82 |
<details>
|
| 83 |
<summary>Show complete input XML</summary>
|
|
@@ -115,62 +114,56 @@ Both channels preserve their shared origin; neither lane is trimmed or shifted.
|
|
| 115 |
|
| 116 |
## Download and run
|
| 117 |
|
| 118 |
-
Use Python 3.12 and a CUDA-capable NVIDIA GPU.
|
| 119 |
-
|
| 120 |
-
|
| 121 |
|
| 122 |
```bash
|
| 123 |
pip install "huggingface_hub>=0.34,<1"
|
| 124 |
-
hf download penguinfish1688/duplexdataengine --local-dir DuDE
|
| 125 |
-
cd DuDE
|
| 126 |
pip install -r requirements.txt
|
| 127 |
python infer.py \
|
| 128 |
-
--model . \
|
| 129 |
-
--xml examples/V01_S0192_I00000307-0000.xml \
|
| 130 |
--voice-a voice_1 --voice-b voice_2 \
|
| 131 |
--seed 20261060 --max-seconds 118 \
|
| 132 |
--output dialogue.wav
|
| 133 |
```
|
| 134 |
|
| 135 |
-
|
| 136 |
|
| 137 |
```python
|
| 138 |
from pathlib import Path
|
| 139 |
from dude_tts import DuDE
|
| 140 |
|
| 141 |
-
|
| 142 |
-
|
| 143 |
-
|
| 144 |
-
|
| 145 |
-
|
| 146 |
)
|
| 147 |
audio.save("dialogue.wav")
|
| 148 |
-
print("A/B emitted EOS:", audio.eos)
|
| 149 |
```
|
| 150 |
|
| 151 |
-
|
| 152 |
-
|
| 153 |
-
|
| 154 |
-
|
| 155 |
-
`--device cpu`, but is
|
| 156 |
|
| 157 |
-
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
|
| 161 |
-
|
| 162 |
-
11.6 GB. The research repository was not installed. See [verification.json](verification.json).
|
| 163 |
|
| 164 |
-
##
|
| 165 |
|
| 166 |
-
Use `<A>` and `<B>` for the
|
| 167 |
-
|
| 168 |
-
|
| 169 |
-
`<e2/>`, and so on. The outer `<duplex>` wrapper is optional.
|
| 170 |
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
Put the anchor at the end of the preceding turn:
|
| 174 |
|
| 175 |
```xml
|
| 176 |
<duplex>
|
|
@@ -180,9 +173,7 @@ Put the anchor at the end of the preceding turn:
|
|
| 180 |
</duplex>
|
| 181 |
```
|
| 182 |
|
| 183 |
-
|
| 184 |
-
|
| 185 |
-
Put the anchor inside a turn to let the other speaker enter there:
|
| 186 |
|
| 187 |
```xml
|
| 188 |
<duplex>
|
|
@@ -192,79 +183,76 @@ Put the anchor inside a turn to let the other speaker enter there:
|
|
| 192 |
</duplex>
|
| 193 |
```
|
| 194 |
|
| 195 |
-
Anchors
|
| 196 |
-
|
| 197 |
-
|
| 198 |
-
PAD. If two turns on one speaker's lane collide, they are serialized in XML
|
| 199 |
-
order; dependent anchors follow the shifted turn. The TTS model predicts the
|
| 200 |
-
audio timing, so exact overlap and duration are not guaranteed.
|
| 201 |
|
| 202 |
-
|
| 203 |
-
`<
|
| 204 |
-
|
| 205 |
-
|
| 206 |
-
timestamp, or duration attributes. Undefined or duplicate anchors, cycles,
|
| 207 |
-
multiple roots, unsupported tags, and XML entity declarations are rejected.
|
| 208 |
|
| 209 |
## Voice templates
|
| 210 |
|
| 211 |
-
|
| 212 |
-
|
| 213 |
-
|
| 214 |
-
|
| 215 |
-
|
| 216 |
-
|
| 217 |
-
|
| 218 |
-
|
| 219 |
-
|
| 220 |
-
|
| 221 |
-
|
| 222 |
-
|
| 223 |
-
|
| 224 |
-
the
|
| 225 |
-
|
| 226 |
-
|
| 227 |
-
|
| 228 |
-
|
| 229 |
-
|
| 230 |
-
|
| 231 |
-
|
| 232 |
-
|
| 233 |
-
|
| 234 |
-
|
| 235 |
-
|
| 236 |
-
|
| 237 |
-
|
| 238 |
-
|
| 239 |
-
|
| 240 |
-
|
| 241 |
-
|
| 242 |
-
|
| 243 |
-
|
| 244 |
-
|
| 245 |
-
|
| 246 |
-
|
| 247 |
-
|
| 248 |
-
|
| 249 |
-
|
| 250 |
-
|
| 251 |
-
|
| 252 |
-
|
| 253 |
-
|
| 254 |
-
|
|
|
|
|
|
|
|
|
|
| 255 |
|
| 256 |
## Citation
|
| 257 |
|
| 258 |
```bibtex
|
| 259 |
@misc{chang2026dude,
|
| 260 |
author = {Cheng-Kuang Chang},
|
| 261 |
-
title = {DuDE:
|
| 262 |
year = {2026},
|
| 263 |
-
howpublished = {Hugging Face model release},
|
| 264 |
url = {https://huggingface.co/penguinfish1688/duplexdataengine}
|
| 265 |
}
|
| 266 |
```
|
| 267 |
|
| 268 |
-
Please also
|
| 269 |
[Seamless Interaction](https://huggingface.co/datasets/facebook/seamless-interaction)
|
| 270 |
-
when
|
|
|
|
| 4 |
license: cc-by-nc-4.0
|
| 5 |
pipeline_tag: text-to-speech
|
| 6 |
tags:
|
| 7 |
+
- full-duplex
|
| 8 |
+
- conversation-tts
|
| 9 |
- dialogue
|
| 10 |
- stereo
|
| 11 |
- xml
|
|
|
|
| 13 |
base_model: Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice
|
| 14 |
datasets:
|
| 15 |
- facebook/seamless-interaction
|
|
|
|
|
|
|
|
|
|
| 16 |
---
|
| 17 |
|
| 18 |
+
# DuDE: Full-Duplex Conversation TTS
|
| 19 |
|
| 20 |
+
DuDE is a full-duplex conversation TTS (text-to-speech) model that generates
|
| 21 |
+
two speakers together from an XML dialogue. It supports turn-taking,
|
| 22 |
+
backchannels, and overlapping speech, with a separate voice for each speaker.
|
| 23 |
+
The output is a 24 kHz stereo WAV: speaker A on the left, speaker B on the right.
|
| 24 |
|
| 25 |
+
## Audio samples
|
| 26 |
|
| 27 |
+
The inputs below are XML transcripts of two held-out conversations from
|
| 28 |
+
[Meta's Seamless Interaction dataset](https://huggingface.co/datasets/facebook/seamless-interaction),
|
| 29 |
+
produced using ASR and vocal-event annotation. **Neither conversation nor any
|
| 30 |
+
of their four speakers appeared in DuDE's training data.** We checked the
|
| 31 |
+
training manifests and retained training caches; see [provenance.json](provenance.json).
|
|
|
|
| 32 |
|
| 33 |
+
The audio is generated by DuDE using speaker reference clips supplied at
|
| 34 |
+
inference time. Both channels are normalized for audibility without shifting
|
| 35 |
+
their shared timeline.
|
| 36 |
|
| 37 |
+
### Sample 1
|
| 38 |
|
| 39 |
+
<audio controls src="https://huggingface.co/penguinfish1688/duplexdataengine/resolve/main/examples/sample_1.wav"></audio>
|
| 40 |
|
| 41 |
+
[Download stereo audio](examples/sample_1.wav) · [Input XML](examples/sample_1.xml) · Voices: `voice_1` / `voice_2`
|
| 42 |
|
| 43 |
<details>
|
| 44 |
<summary>Show complete input XML</summary>
|
|
|
|
| 72 |
|
| 73 |
</details>
|
| 74 |
|
| 75 |
+
### Sample 2
|
| 76 |
|
| 77 |
+
<audio controls src="https://huggingface.co/penguinfish1688/duplexdataengine/resolve/main/examples/sample_2.wav"></audio>
|
| 78 |
|
| 79 |
+
[Download stereo audio](examples/sample_2.wav) · [Input XML](examples/sample_2.xml) · Voices: `voice_3` / `voice_4`
|
| 80 |
|
| 81 |
<details>
|
| 82 |
<summary>Show complete input XML</summary>
|
|
|
|
| 114 |
|
| 115 |
## Download and run
|
| 116 |
|
| 117 |
+
Use Python 3.12 and a CUDA-capable NVIDIA GPU. The download includes the model,
|
| 118 |
+
audio codec, XML frontend, four voice templates, and inference code. Requirements
|
| 119 |
+
pin PyTorch and torchaudio 2.8.0; use the corresponding CUDA builds for your system.
|
| 120 |
|
| 121 |
```bash
|
| 122 |
pip install "huggingface_hub>=0.34,<1"
|
| 123 |
+
hf download penguinfish1688/duplexdataengine --local-dir DuDE
|
| 124 |
+
cd DuDE
|
| 125 |
pip install -r requirements.txt
|
| 126 |
python infer.py \
|
| 127 |
+
--model . --xml examples/sample_1.xml \
|
|
|
|
| 128 |
--voice-a voice_1 --voice-b voice_2 \
|
| 129 |
--seed 20261060 --max-seconds 118 \
|
| 130 |
--output dialogue.wav
|
| 131 |
```
|
| 132 |
|
| 133 |
+
Or use the Python API from the downloaded directory:
|
| 134 |
|
| 135 |
```python
|
| 136 |
from pathlib import Path
|
| 137 |
from dude_tts import DuDE
|
| 138 |
|
| 139 |
+
model = DuDE.from_pretrained(".", device="cuda")
|
| 140 |
+
audio = model.generate(
|
| 141 |
+
Path("examples/sample_1.xml").read_text(),
|
| 142 |
+
voice_a="voice_1", voice_b="voice_2",
|
| 143 |
+
seed=20261060, max_seconds=118,
|
| 144 |
)
|
| 145 |
audio.save("dialogue.wav")
|
|
|
|
| 146 |
```
|
| 147 |
|
| 148 |
+
`max_seconds` is an output limit, not a target speaking duration. Generation
|
| 149 |
+
finishes when both speakers emit an end-of-speech token; `audio.eos` reports
|
| 150 |
+
whether each speaker finished. Leading silence and pauses are preserved, with
|
| 151 |
+
silence padding after the shorter channel ends. CPU inference is available
|
| 152 |
+
with `--device cpu`, but is slow.
|
| 153 |
|
| 154 |
+
An anonymous download was tested in a fresh environment on one RTX PRO 6000
|
| 155 |
+
Blackwell with PyTorch 2.8.0+cu128. Generating 47.6 and 54.4 seconds of stereo
|
| 156 |
+
audio took 50.7 and 57.4 seconds, respectively, plus about 28 seconds to load
|
| 157 |
+
the model. Peak allocated GPU memory was 11.6 GB. Both speakers finished in
|
| 158 |
+
both samples. See [verification.json](verification.json).
|
|
|
|
| 159 |
|
| 160 |
+
## Writing a dialogue
|
| 161 |
|
| 162 |
+
Use `<A>` and `<B>` for the speakers. Exactly one initial turn has no `start`
|
| 163 |
+
attribute. Every later turn references an inline anchor with `start="e1"`,
|
| 164 |
+
`start="e2"`, and so on. Define each anchor once. The `<duplex>` wrapper is optional.
|
|
|
|
| 165 |
|
| 166 |
+
For turn-taking, put the anchor at the preceding turn's end:
|
|
|
|
|
|
|
| 167 |
|
| 168 |
```xml
|
| 169 |
<duplex>
|
|
|
|
| 173 |
</duplex>
|
| 174 |
```
|
| 175 |
|
| 176 |
+
For overlap, put the anchor where the other speaker should enter:
|
|
|
|
|
|
|
| 177 |
|
| 178 |
```xml
|
| 179 |
<duplex>
|
|
|
|
| 183 |
</duplex>
|
| 184 |
```
|
| 185 |
|
| 186 |
+
Anchors express turn relationships; the model predicts the audio timing.
|
| 187 |
+
Exact entry times are not guaranteed. Colliding turns for the same speaker
|
| 188 |
+
follow their XML order. All references must exist, and turns must not form a cycle.
|
|
|
|
|
|
|
|
|
|
| 189 |
|
| 190 |
+
Vocal events: `<laugh/>`, `<cough/>`, `<breath/>`, `<sigh/>`, `<cry/>`,
|
| 191 |
+
`<sneeze/>`, and `<gasp/>`. Write spoken backchannels as ordinary text and
|
| 192 |
+
escape literal ampersands as `&`. Attributes such as `type`, timestamps,
|
| 193 |
+
and duration are not supported.
|
|
|
|
|
|
|
| 194 |
|
| 195 |
## Voice templates
|
| 196 |
|
| 197 |
+
Choose a template independently for each speaker. The current API supports
|
| 198 |
+
these four presets; it does not accept uploaded voice clips.
|
| 199 |
+
|
| 200 |
+
| Template | Reference audio |
|
| 201 |
+
| --- | --- |
|
| 202 |
+
| `voice_1` | [Sample 1, speaker A](voices/voice_1.wav) |
|
| 203 |
+
| `voice_2` | [Sample 1, speaker B](voices/voice_2.wav) |
|
| 204 |
+
| `voice_3` | [Sample 2, speaker A](voices/voice_3.wav) |
|
| 205 |
+
| `voice_4` | [Sample 2, speaker B](voices/voice_4.wav) |
|
| 206 |
+
|
| 207 |
+
## Technical details
|
| 208 |
+
|
| 209 |
+
DuDE uses a shared Qwen3-TTS backbone for both speakers. A deterministic
|
| 210 |
+
frontend resolves the XML turn dependencies and densely interleaves the two
|
| 211 |
+
text channels. Voice embeddings condition each speaker, and the model generates
|
| 212 |
+
both audio channels autoregressively with independent end-of-speech decisions.
|
| 213 |
+
The codec decodes them onto a shared timeline.
|
| 214 |
+
|
| 215 |
+
The model was trained on approximately 1,922 hours of synchronized Seamless
|
| 216 |
+
Interaction conversations. The release contains merged model weights in FP32;
|
| 217 |
+
CUDA inference uses BF16 autocast.
|
| 218 |
+
|
| 219 |
+
## Evaluation
|
| 220 |
+
|
| 221 |
+
On ten held-out English conversations, both speakers finished in every example.
|
| 222 |
+
Automated word error rate was **15.70%** for generated speech and **12.65%**
|
| 223 |
+
for an ASR transcription of the original recordings. This is a small evaluation
|
| 224 |
+
set, without a human listening study.
|
| 225 |
+
|
| 226 |
+
Timing is less reliable: lexical timing coverage was **44.79%**, with **72.09%**
|
| 227 |
+
agreement among measured timing constraints. The timing check did not meet its
|
| 228 |
+
acceptance criteria. Outputs can contain wrong words, long pauses, missed vocal
|
| 229 |
+
events, or inaccurate overlaps. Other languages and individual vocal events
|
| 230 |
+
have not been separately evaluated.
|
| 231 |
+
|
| 232 |
+
## License
|
| 233 |
+
|
| 234 |
+
Model weights, voice references, and conversation examples are available for
|
| 235 |
+
noncommercial use under [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/).
|
| 236 |
+
DuDE inference code is Apache-2.0; see [LICENSE-CODE](LICENSE-CODE).
|
| 237 |
+
Qwen components retain their Apache-2.0 notices in [LICENSE-QWEN](LICENSE-QWEN).
|
| 238 |
+
|
| 239 |
+
Conversation material and reference voices derive from Meta's
|
| 240 |
+
[Seamless Interaction dataset](https://huggingface.co/datasets/facebook/seamless-interaction),
|
| 241 |
+
also licensed CC BY-NC 4.0. This release includes derived transcripts, XML,
|
| 242 |
+
speaker embeddings, and synthesized recordings. Preserve the dataset attribution
|
| 243 |
+
when redistributing these assets.
|
| 244 |
|
| 245 |
## Citation
|
| 246 |
|
| 247 |
```bibtex
|
| 248 |
@misc{chang2026dude,
|
| 249 |
author = {Cheng-Kuang Chang},
|
| 250 |
+
title = {DuDE: Full-Duplex Conversation TTS},
|
| 251 |
year = {2026},
|
|
|
|
| 252 |
url = {https://huggingface.co/penguinfish1688/duplexdataengine}
|
| 253 |
}
|
| 254 |
```
|
| 255 |
|
| 256 |
+
Please also cite [Qwen3-TTS](https://github.com/QwenLM/Qwen3-TTS) and
|
| 257 |
[Seamless Interaction](https://huggingface.co/datasets/facebook/seamless-interaction)
|
| 258 |
+
when using this model.
|
dude_tts/interleave.py
CHANGED
|
@@ -1,4 +1,4 @@
|
|
| 1 |
-
"""
|
| 2 |
|
| 3 |
No timestamps or XML tokens enter the lanes. Start anchors are token boundaries.
|
| 4 |
Same-channel turns are serialized in XML order; a delayed turn carries all of
|
|
|
|
| 1 |
+
"""Dialogue text frontend: XML -> dependency DAG -> dense, ordered lanes.
|
| 2 |
|
| 3 |
No timestamps or XML tokens enter the lanes. Start anchors are token boundaries.
|
| 4 |
Same-channel turns are serialized in XML order; a delayed turn carries all of
|
examples/examples.json
CHANGED
|
@@ -1,20 +1,24 @@
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
-
"id": "
|
| 4 |
-
"xml": "examples/
|
| 5 |
-
"audio": "examples/
|
| 6 |
"voice_a": "voice_1",
|
| 7 |
"voice_b": "voice_2",
|
| 8 |
"seed": 20261060,
|
| 9 |
-
"max_seconds": 118.0
|
|
|
|
|
|
|
| 10 |
},
|
| 11 |
{
|
| 12 |
-
"id": "
|
| 13 |
-
"xml": "examples/
|
| 14 |
-
"audio": "examples/
|
| 15 |
"voice_a": "voice_3",
|
| 16 |
"voice_b": "voice_4",
|
| 17 |
"seed": 20261063,
|
| 18 |
-
"max_seconds": 134.0
|
|
|
|
|
|
|
| 19 |
}
|
| 20 |
]
|
|
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
+
"id": "sample_1",
|
| 4 |
+
"xml": "examples/sample_1.xml",
|
| 5 |
+
"audio": "examples/sample_1.wav",
|
| 6 |
"voice_a": "voice_1",
|
| 7 |
"voice_b": "voice_2",
|
| 8 |
"seed": 20261060,
|
| 9 |
+
"max_seconds": 118.0,
|
| 10 |
+
"label": "Sample 1",
|
| 11 |
+
"source_conversation": "V01_S0192_I00000307"
|
| 12 |
},
|
| 13 |
{
|
| 14 |
+
"id": "sample_2",
|
| 15 |
+
"xml": "examples/sample_2.xml",
|
| 16 |
+
"audio": "examples/sample_2.wav",
|
| 17 |
"voice_a": "voice_3",
|
| 18 |
"voice_b": "voice_4",
|
| 19 |
"seed": 20261063,
|
| 20 |
+
"max_seconds": 134.0,
|
| 21 |
+
"label": "Sample 2",
|
| 22 |
+
"source_conversation": "V01_S0220_I00000126"
|
| 23 |
}
|
| 24 |
]
|
examples/{V01_S0192_I00000307-0000-student.wav → sample_1.wav}
RENAMED
|
File without changes
|
examples/{V01_S0192_I00000307-0000.xml → sample_1.xml}
RENAMED
|
File without changes
|
examples/{V01_S0220_I00000126-0000-student.wav → sample_2.wav}
RENAMED
|
File without changes
|
examples/{V01_S0220_I00000126-0000.xml → sample_2.xml}
RENAMED
|
File without changes
|
provenance.json
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"passed": true,
|
| 3 |
+
"checked_at_utc": "2026-09-27T02:40:28.055408+00:00",
|
| 4 |
+
"scope": "DuDE training data; upstream Qwen pretraining coverage is not known",
|
| 5 |
+
"examples": [
|
| 6 |
+
{
|
| 7 |
+
"label": "Sample 1",
|
| 8 |
+
"conversation_id": "V01_S0192_I00000307",
|
| 9 |
+
"example_id": "V01_S0192_I00000307-0000",
|
| 10 |
+
"split": "dev",
|
| 11 |
+
"participants": {
|
| 12 |
+
"A": "V01_P1356",
|
| 13 |
+
"B": "V01_P1357"
|
| 14 |
+
},
|
| 15 |
+
"transcript_source": "own_Qwen3-ASR+ForcedAligner+PANNs",
|
| 16 |
+
"dataset_transcript_downloaded": false,
|
| 17 |
+
"annotation_sha256": "76474198f55a6d7afd7f826cd910b9841af8293a32413f2f8b672c1ef7fd4996",
|
| 18 |
+
"public_example_id": "sample_1"
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"label": "Sample 2",
|
| 22 |
+
"conversation_id": "V01_S0220_I00000126",
|
| 23 |
+
"example_id": "V01_S0220_I00000126-0000",
|
| 24 |
+
"split": "dev",
|
| 25 |
+
"participants": {
|
| 26 |
+
"A": "V01_P1409",
|
| 27 |
+
"B": "V01_P1500"
|
| 28 |
+
},
|
| 29 |
+
"transcript_source": "own_Qwen3-ASR+ForcedAligner+PANNs",
|
| 30 |
+
"dataset_transcript_downloaded": false,
|
| 31 |
+
"annotation_sha256": "d3d434ee2d09aa3916f1883df7e1b1134125b285a2a09cac64171c49110d1cb6",
|
| 32 |
+
"public_example_id": "sample_2"
|
| 33 |
+
}
|
| 34 |
+
],
|
| 35 |
+
"training_manifests": [
|
| 36 |
+
{
|
| 37 |
+
"corpus": "seamless-100h",
|
| 38 |
+
"train_conversations": 1468,
|
| 39 |
+
"sha256": "05c7e77c69078b9d68ef3c30a68392d893f82f7a5cca94bdbf2e3a4ced40d282",
|
| 40 |
+
"conversation_matches": [],
|
| 41 |
+
"participant_matches": [],
|
| 42 |
+
"featured_conversations_in_dev": [
|
| 43 |
+
"V01_S0220_I00000126"
|
| 44 |
+
]
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"corpus": "seamless-500h-v1",
|
| 48 |
+
"train_conversations": 7365,
|
| 49 |
+
"sha256": "2967b0c49574182fd14d56571107dbc264f199d6cd3ed39ea71bc9956a81203a",
|
| 50 |
+
"conversation_matches": [],
|
| 51 |
+
"participant_matches": [],
|
| 52 |
+
"featured_conversations_in_dev": [
|
| 53 |
+
"V01_S0192_I00000307",
|
| 54 |
+
"V01_S0220_I00000126"
|
| 55 |
+
]
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"corpus": "seamless-2000h-v1",
|
| 59 |
+
"train_conversations": 31493,
|
| 60 |
+
"sha256": "08d437e07c00ebd4a73feed219e0048e69d1bdcb681dac12f20b0fd9c803758e",
|
| 61 |
+
"conversation_matches": [],
|
| 62 |
+
"participant_matches": [],
|
| 63 |
+
"featured_conversations_in_dev": [
|
| 64 |
+
"V01_S0192_I00000307",
|
| 65 |
+
"V01_S0220_I00000126"
|
| 66 |
+
]
|
| 67 |
+
}
|
| 68 |
+
],
|
| 69 |
+
"training_codec_caches": [
|
| 70 |
+
{
|
| 71 |
+
"corpus": "conversation-2000h-v1",
|
| 72 |
+
"shards": 8,
|
| 73 |
+
"examples": 30854,
|
| 74 |
+
"featured_conversation_matches": []
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
"corpus": "conversation-500h-full-v1",
|
| 78 |
+
"shards": 4,
|
| 79 |
+
"examples": 7190,
|
| 80 |
+
"featured_conversation_matches": []
|
| 81 |
+
}
|
| 82 |
+
],
|
| 83 |
+
"transcript_note": "XML uses our ASR/event annotations of held-out Seamless Interaction audio, not dataset-provided transcripts.",
|
| 84 |
+
"voice_reference_note": "Speaker reference clips are provided during inference; they are not training examples."
|
| 85 |
+
}
|
release.json
CHANGED
|
@@ -1,7 +1,6 @@
|
|
| 1 |
{
|
| 2 |
-
"format": "
|
| 3 |
"version": "1.0.0",
|
| 4 |
-
"checkpoint_step": 8565,
|
| 5 |
"author": "Cheng-Kuang Chang",
|
| 6 |
"base_model": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice",
|
| 7 |
"base_revision": "0c0e3051f131929182e2c023b9537f8b1c68adfe",
|
|
@@ -25,22 +24,26 @@
|
|
| 25 |
},
|
| 26 |
"examples": [
|
| 27 |
{
|
| 28 |
-
"id": "
|
| 29 |
-
"xml": "examples/
|
| 30 |
-
"audio": "examples/
|
| 31 |
"voice_a": "voice_1",
|
| 32 |
"voice_b": "voice_2",
|
| 33 |
"seed": 20261060,
|
| 34 |
-
"max_seconds": 118.0
|
|
|
|
|
|
|
| 35 |
},
|
| 36 |
{
|
| 37 |
-
"id": "
|
| 38 |
-
"xml": "examples/
|
| 39 |
-
"audio": "examples/
|
| 40 |
"voice_a": "voice_3",
|
| 41 |
"voice_b": "voice_4",
|
| 42 |
"seed": 20261063,
|
| 43 |
-
"max_seconds": 134.0
|
|
|
|
|
|
|
| 44 |
}
|
| 45 |
],
|
| 46 |
"license": "cc-by-nc-4.0"
|
|
|
|
| 1 |
{
|
| 2 |
+
"format": "dude_duplex_v1",
|
| 3 |
"version": "1.0.0",
|
|
|
|
| 4 |
"author": "Cheng-Kuang Chang",
|
| 5 |
"base_model": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice",
|
| 6 |
"base_revision": "0c0e3051f131929182e2c023b9537f8b1c68adfe",
|
|
|
|
| 24 |
},
|
| 25 |
"examples": [
|
| 26 |
{
|
| 27 |
+
"id": "sample_1",
|
| 28 |
+
"xml": "examples/sample_1.xml",
|
| 29 |
+
"audio": "examples/sample_1.wav",
|
| 30 |
"voice_a": "voice_1",
|
| 31 |
"voice_b": "voice_2",
|
| 32 |
"seed": 20261060,
|
| 33 |
+
"max_seconds": 118.0,
|
| 34 |
+
"label": "Sample 1",
|
| 35 |
+
"source_conversation": "V01_S0192_I00000307"
|
| 36 |
},
|
| 37 |
{
|
| 38 |
+
"id": "sample_2",
|
| 39 |
+
"xml": "examples/sample_2.xml",
|
| 40 |
+
"audio": "examples/sample_2.wav",
|
| 41 |
"voice_a": "voice_3",
|
| 42 |
"voice_b": "voice_4",
|
| 43 |
"seed": 20261063,
|
| 44 |
+
"max_seconds": 134.0,
|
| 45 |
+
"label": "Sample 2",
|
| 46 |
+
"source_conversation": "V01_S0220_I00000126"
|
| 47 |
}
|
| 48 |
],
|
| 49 |
"license": "cc-by-nc-4.0"
|
verification.json
CHANGED
|
@@ -5,12 +5,11 @@
|
|
| 5 |
"anonymous_download": true,
|
| 6 |
"isolated_environment": true,
|
| 7 |
"research_checkout_required": false,
|
| 8 |
-
"source_checkpoint_step": 8565,
|
| 9 |
"gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
|
| 10 |
"torch": "2.8.0+cu128",
|
| 11 |
"examples": [
|
| 12 |
{
|
| 13 |
-
"id": "
|
| 14 |
"eos": [
|
| 15 |
true,
|
| 16 |
true
|
|
@@ -25,10 +24,11 @@
|
|
| 25 |
"correlation_with_featured_audio": [
|
| 26 |
0.9999125589163387,
|
| 27 |
0.9996439714705407
|
| 28 |
-
]
|
|
|
|
| 29 |
},
|
| 30 |
{
|
| 31 |
-
"id": "
|
| 32 |
"eos": [
|
| 33 |
true,
|
| 34 |
true
|
|
@@ -43,7 +43,8 @@
|
|
| 43 |
"correlation_with_featured_audio": [
|
| 44 |
0.9999395525329423,
|
| 45 |
0.9998978182617578
|
| 46 |
-
]
|
|
|
|
| 47 |
}
|
| 48 |
],
|
| 49 |
"peak_gpu_memory_gb": 11.553627648,
|
|
@@ -58,5 +59,6 @@
|
|
| 58 |
"Both lanes emit EOS",
|
| 59 |
"Finite normalized 24 kHz stereo waveform",
|
| 60 |
"Shared clock preserved; waveform correlation above 0.99 with each featured lane"
|
| 61 |
-
]
|
|
|
|
| 62 |
}
|
|
|
|
| 5 |
"anonymous_download": true,
|
| 6 |
"isolated_environment": true,
|
| 7 |
"research_checkout_required": false,
|
|
|
|
| 8 |
"gpu": "NVIDIA RTX PRO 6000 Blackwell Server Edition",
|
| 9 |
"torch": "2.8.0+cu128",
|
| 10 |
"examples": [
|
| 11 |
{
|
| 12 |
+
"id": "sample_1",
|
| 13 |
"eos": [
|
| 14 |
true,
|
| 15 |
true
|
|
|
|
| 24 |
"correlation_with_featured_audio": [
|
| 25 |
0.9999125589163387,
|
| 26 |
0.9996439714705407
|
| 27 |
+
],
|
| 28 |
+
"label": "Sample 1"
|
| 29 |
},
|
| 30 |
{
|
| 31 |
+
"id": "sample_2",
|
| 32 |
"eos": [
|
| 33 |
true,
|
| 34 |
true
|
|
|
|
| 43 |
"correlation_with_featured_audio": [
|
| 44 |
0.9999395525329423,
|
| 45 |
0.9998978182617578
|
| 46 |
+
],
|
| 47 |
+
"label": "Sample 2"
|
| 48 |
}
|
| 49 |
],
|
| 50 |
"peak_gpu_memory_gb": 11.553627648,
|
|
|
|
| 59 |
"Both lanes emit EOS",
|
| 60 |
"Finite normalized 24 kHz stereo waveform",
|
| 61 |
"Shared clock preserved; waveform correlation above 0.99 with each featured lane"
|
| 62 |
+
],
|
| 63 |
+
"model_version": "1.0.0"
|
| 64 |
}
|
voices/presets.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
{
|
| 2 |
"voice_1": {
|
| 3 |
-
"label": "
|
| 4 |
"embedding": [
|
| 5 |
0.0054931640625,
|
| 6 |
0.0274658203125,
|
|
@@ -2060,7 +2060,7 @@
|
|
| 2060 |
"license": "cc-by-nc-4.0"
|
| 2061 |
},
|
| 2062 |
"voice_2": {
|
| 2063 |
-
"label": "
|
| 2064 |
"embedding": [
|
| 2065 |
0.103515625,
|
| 2066 |
-0.007080078125,
|
|
@@ -4120,7 +4120,7 @@
|
|
| 4120 |
"license": "cc-by-nc-4.0"
|
| 4121 |
},
|
| 4122 |
"voice_3": {
|
| 4123 |
-
"label": "
|
| 4124 |
"embedding": [
|
| 4125 |
-0.0400390625,
|
| 4126 |
0.0390625,
|
|
@@ -6180,7 +6180,7 @@
|
|
| 6180 |
"license": "cc-by-nc-4.0"
|
| 6181 |
},
|
| 6182 |
"voice_4": {
|
| 6183 |
-
"label": "
|
| 6184 |
"embedding": [
|
| 6185 |
0.0634765625,
|
| 6186 |
-0.058837890625,
|
|
|
|
| 1 |
{
|
| 2 |
"voice_1": {
|
| 3 |
+
"label": "Sample 1 \u00b7 speaker A",
|
| 4 |
"embedding": [
|
| 5 |
0.0054931640625,
|
| 6 |
0.0274658203125,
|
|
|
|
| 2060 |
"license": "cc-by-nc-4.0"
|
| 2061 |
},
|
| 2062 |
"voice_2": {
|
| 2063 |
+
"label": "Sample 1 \u00b7 speaker B",
|
| 2064 |
"embedding": [
|
| 2065 |
0.103515625,
|
| 2066 |
-0.007080078125,
|
|
|
|
| 4120 |
"license": "cc-by-nc-4.0"
|
| 4121 |
},
|
| 4122 |
"voice_3": {
|
| 4123 |
+
"label": "Sample 2 \u00b7 speaker A",
|
| 4124 |
"embedding": [
|
| 4125 |
-0.0400390625,
|
| 4126 |
0.0390625,
|
|
|
|
| 6180 |
"license": "cc-by-nc-4.0"
|
| 6181 |
},
|
| 6182 |
"voice_4": {
|
| 6183 |
+
"label": "Sample 2 \u00b7 speaker B",
|
| 6184 |
"embedding": [
|
| 6185 |
0.0634765625,
|
| 6186 |
-0.058837890625,
|