caiyiyi1998 commited on
Commit
2402033
·
0 Parent(s):

Initial commit

Browse files
Files changed (44) hide show
  1. .gitattributes +35 -0
  2. README.md +77 -0
  3. assets_manifest.json +327 -0
  4. checkpoints/humanml3d_babel_fk_200k/assets/Mean.npy +3 -0
  5. checkpoints/humanml3d_babel_fk_200k/assets/Std.npy +3 -0
  6. checkpoints/humanml3d_babel_fk_200k/assets/W.npy +3 -0
  7. checkpoints/humanml3d_babel_fk_200k/config.yaml +165 -0
  8. checkpoints/humanml3d_babel_fk_200k/model.ckpt +3 -0
  9. checkpoints/humanml3d_babel_path_200k/assets/Mean.npy +3 -0
  10. checkpoints/humanml3d_babel_path_200k/assets/Std.npy +3 -0
  11. checkpoints/humanml3d_babel_path_200k/assets/W.npy +3 -0
  12. checkpoints/humanml3d_babel_path_200k/config.yaml +167 -0
  13. checkpoints/humanml3d_babel_path_200k/model.ckpt +3 -0
  14. checkpoints/humanml3d_fk_60k/assets/Mean.npy +3 -0
  15. checkpoints/humanml3d_fk_60k/assets/Std.npy +3 -0
  16. checkpoints/humanml3d_fk_60k/assets/W.npy +3 -0
  17. checkpoints/humanml3d_fk_60k/config.yaml +144 -0
  18. checkpoints/humanml3d_fk_60k/model.ckpt +3 -0
  19. checkpoints/humanml3d_path_fk_55k/assets/Mean.npy +3 -0
  20. checkpoints/humanml3d_path_fk_55k/assets/Std.npy +3 -0
  21. checkpoints/humanml3d_path_fk_55k/assets/W.npy +3 -0
  22. checkpoints/humanml3d_path_fk_55k/config.yaml +145 -0
  23. checkpoints/humanml3d_path_fk_55k/model.ckpt +3 -0
  24. checkpoints/ldf_263/assets/Mean.npy +3 -0
  25. checkpoints/ldf_263/assets/Std.npy +3 -0
  26. checkpoints/ldf_263/config.yaml +174 -0
  27. checkpoints/ldf_263/model.ckpt +3 -0
  28. checkpoints/seed_fk_300k/assets/Mean.npy +3 -0
  29. checkpoints/seed_fk_300k/assets/Std.npy +3 -0
  30. checkpoints/seed_fk_300k/assets/W.npy +3 -0
  31. checkpoints/seed_fk_300k/config.yaml +143 -0
  32. checkpoints/seed_fk_300k/model.ckpt +3 -0
  33. checkpoints/seed_path_fk_300k/assets/Mean.npy +3 -0
  34. checkpoints/seed_path_fk_300k/assets/Std.npy +3 -0
  35. checkpoints/seed_path_fk_300k/assets/W.npy +3 -0
  36. checkpoints/seed_path_fk_300k/config.yaml +145 -0
  37. checkpoints/seed_path_fk_300k/model.ckpt +3 -0
  38. checkpoints/vae_263/assets/Mean.npy +3 -0
  39. checkpoints/vae_263/assets/Std.npy +3 -0
  40. checkpoints/vae_263/config.yaml +104 -0
  41. checkpoints/vae_263/model.ckpt +3 -0
  42. dependencies.json +74 -0
  43. deps.zip +3 -0
  44. publication_manifest.json +402 -0
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: pytorch
3
+ tags:
4
+ - motion-generation
5
+ - text-to-motion
6
+ - diffusion
7
+ - streaming
8
+ - humanml3d
9
+ language:
10
+ - en
11
+ ---
12
+
13
+ # FloodDiffusion 2
14
+
15
+ **Efficient and Path Controllable Streaming Motion Generation**
16
+
17
+ [Code and setup instructions](https://github.com/AlayaLab/FloodDiffusion2)
18
+
19
+ FloodDiffusion 2 supports continuous text-conditioned motion generation with
20
+ Partial Attention, diffusion-compatible geometric supervision, and root-path
21
+ control. This release includes six FD2 checkpoints and the paired 263D LDF/VAE
22
+ models for first-generation compatibility.
23
+
24
+ | Checkpoint folder | Configuration | Training step | CFG |
25
+ |---|---|---:|---:|
26
+ | `humanml3d_fk_60k` | `df_humanml3d_263.yaml` | 60,000 | 4 |
27
+ | `humanml3d_path_fk_55k` | `df_humanml3d_263_path.yaml` | 55,000 | 3 |
28
+ | `humanml3d_babel_fk_200k` | `df_humanml3d_babel_263.yaml` | 200,000 | 4 |
29
+ | `humanml3d_babel_path_200k` | `df_humanml3d_babel_263_path.yaml` | 200,000 | 4 |
30
+ | `seed_fk_300k` | `df_seed_138.yaml` | 300,000 | 2 |
31
+ | `seed_path_fk_300k` | `df_seed_138_path.yaml` | 300,000 | 2 |
32
+ | `ldf_263` | `ldf_263.yaml` | 180,000 | 6 |
33
+ | `vae_263` | `vae_263.yaml` | 2,250,000 | — |
34
+
35
+ Each `checkpoints/<folder>/` contains `model.ckpt`, a matching `config.yaml`,
36
+ and normalization statistics under `assets/`. FD2 folders also include the
37
+ quadratic FK matrix `assets/W.npy`. Checkpoints retain model, EMA, optimizer and
38
+ scheduler state.
39
+
40
+ ## Setup
41
+
42
+ Clone the code repository and run the following in a Python 3.10+ environment:
43
+
44
+ ```bash
45
+ python setup_project.py
46
+ ```
47
+
48
+ The command installs Python requirements and downloads the four paper checkpoints
49
+ (HumanML3D text/path and SEED text/path), their assets, and the
50
+ [shared dependency archive](https://huggingface.co/AlayaLab/FloodDiffusion2/blob/main/deps.zip)
51
+ for UMT5, T2M and GloVe. Add `--with-data` to also install the
52
+ [prepared datasets](https://huggingface.co/datasets/AlayaLab/FloodDiffusion2-Data)
53
+ and paired VAE tokens.
54
+
55
+ For mesh rendering, obtain SMPL-H from its
56
+ [official site](https://mano.is.tue.mpg.de/) and import the licensed neutral model:
57
+
58
+ ```bash
59
+ python setup_project.py --smplh /path/to/neutral/model.npz
60
+ ```
61
+
62
+ ## Evaluation
63
+
64
+ ```bash
65
+ python evaluate.py --config configs/df_humanml3d_263.yaml
66
+ ```
67
+
68
+ For additional checkpoints, use their bundled `checkpoints/<folder>/config.yaml`.
69
+
70
+ The selected YAML defines the checkpoint, dataset and evaluation settings.
71
+ Use the corresponding configuration for other variants. LDF uses the paired
72
+ VAE from this release.
73
+
74
+ ## Dependencies
75
+
76
+ Third-party weights and datasets retain their respective licenses. SMPL-H is
77
+ obtained separately from its official distributor.
assets_manifest.json ADDED
@@ -0,0 +1,327 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format_version": 1,
3
+ "repository": "AlayaLab/FloodDiffusion2",
4
+ "revision": "alaya-initial",
5
+ "models": [
6
+ {
7
+ "name": "vae_263",
8
+ "config": "configs/vae_263.yaml",
9
+ "step": 2250000,
10
+ "files": [
11
+ {
12
+ "path": "checkpoints/vae_263/model.ckpt",
13
+ "size": 350279555,
14
+ "sha256": "b80ce00b538b24b874dd15ac84fa4f34d9469fe7ac7f8d99e60bea276c3918d4"
15
+ },
16
+ {
17
+ "path": "checkpoints/vae_263/config.yaml",
18
+ "size": 2581,
19
+ "sha256": "7047f20640a97538d8629bf91cd12c667d7992740894e051501eb1badd0f098e"
20
+ },
21
+ {
22
+ "path": "checkpoints/vae_263/assets/Mean.npy",
23
+ "size": 1180,
24
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
25
+ },
26
+ {
27
+ "path": "checkpoints/vae_263/assets/Std.npy",
28
+ "size": 1180,
29
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
30
+ }
31
+ ]
32
+ },
33
+ {
34
+ "name": "ldf_263",
35
+ "config": "configs/ldf_263.yaml",
36
+ "step": 180000,
37
+ "files": [
38
+ {
39
+ "path": "checkpoints/ldf_263/model.ckpt",
40
+ "size": 2274793067,
41
+ "sha256": "94cc2ea9c0c29213a7f615e721c5fe5264e96d0a95f7f67adc0a5a5081f69437"
42
+ },
43
+ {
44
+ "path": "checkpoints/ldf_263/config.yaml",
45
+ "size": 4564,
46
+ "sha256": "4a252ff70dbd9a1864a9f0afb7e71f0842440d1cc8bcdf360ab1769cbcf65f41"
47
+ },
48
+ {
49
+ "path": "checkpoints/ldf_263/assets/Mean.npy",
50
+ "size": 144,
51
+ "sha256": "c8b1c3183972f4ba2e542eb8bc26945f5d13782d04b56345d6ab5c71e714dff6"
52
+ },
53
+ {
54
+ "path": "checkpoints/ldf_263/assets/Std.npy",
55
+ "size": 144,
56
+ "sha256": "a4234b3c2ca19566cd74b749e1c1ede5cbc2ab245cfcc487891b14c056a2357d"
57
+ }
58
+ ]
59
+ },
60
+ {
61
+ "name": "humanml3d_fk_60k",
62
+ "config": "configs/df_humanml3d_263.yaml",
63
+ "step": 60000,
64
+ "files": [
65
+ {
66
+ "path": "checkpoints/humanml3d_fk_60k/model.ckpt",
67
+ "size": 2285681258,
68
+ "sha256": "d48332bcd88d06098627ed895e8d708702290d6ed518b20e5573b6b959725a63"
69
+ },
70
+ {
71
+ "path": "checkpoints/humanml3d_fk_60k/config.yaml",
72
+ "size": 3811,
73
+ "sha256": "983449db24beb87f9d63fbde21927e7119990bf10d6af6e32324f1a5526a7b8f"
74
+ },
75
+ {
76
+ "path": "checkpoints/humanml3d_fk_60k/assets/Mean.npy",
77
+ "size": 1180,
78
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
79
+ },
80
+ {
81
+ "path": "checkpoints/humanml3d_fk_60k/assets/Std.npy",
82
+ "size": 1180,
83
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
84
+ },
85
+ {
86
+ "path": "checkpoints/humanml3d_fk_60k/assets/W.npy",
87
+ "size": 553480,
88
+ "sha256": "4a8411b10882a74d09479e20e8ec10d0208473127f23e9da4d3192e2fea145d6"
89
+ }
90
+ ]
91
+ },
92
+ {
93
+ "name": "humanml3d_path_fk_55k",
94
+ "config": "configs/df_humanml3d_263_path.yaml",
95
+ "step": 55000,
96
+ "files": [
97
+ {
98
+ "path": "checkpoints/humanml3d_path_fk_55k/model.ckpt",
99
+ "size": 2285614312,
100
+ "sha256": "ee1239ee5458781ab84d21329d190bfcb45a23d12bec17df48924e000672d104"
101
+ },
102
+ {
103
+ "path": "checkpoints/humanml3d_path_fk_55k/config.yaml",
104
+ "size": 3877,
105
+ "sha256": "efcb66b0ec75c721b418887fa9824285dcc8934dc98550218348303bda273332"
106
+ },
107
+ {
108
+ "path": "checkpoints/humanml3d_path_fk_55k/assets/Mean.npy",
109
+ "size": 1180,
110
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
111
+ },
112
+ {
113
+ "path": "checkpoints/humanml3d_path_fk_55k/assets/Std.npy",
114
+ "size": 1180,
115
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
116
+ },
117
+ {
118
+ "path": "checkpoints/humanml3d_path_fk_55k/assets/W.npy",
119
+ "size": 540928,
120
+ "sha256": "ebc7243c5185ab6203af8e0b912e05d6bd15919f9eda7391b548a870eefea373"
121
+ }
122
+ ]
123
+ },
124
+ {
125
+ "name": "humanml3d_babel_fk_200k",
126
+ "config": "configs/df_humanml3d_babel_263.yaml",
127
+ "step": 200000,
128
+ "files": [
129
+ {
130
+ "path": "checkpoints/humanml3d_babel_fk_200k/model.ckpt",
131
+ "size": 2285685418,
132
+ "sha256": "216a6168daca6552495ff8415a7cc31d896e9dfb116e4165e1b777ac609e6ae3"
133
+ },
134
+ {
135
+ "path": "checkpoints/humanml3d_babel_fk_200k/config.yaml",
136
+ "size": 4434,
137
+ "sha256": "69ee7e4642b5baa3bba4e9e89e14a69c020cb936075ced8683be98426488eac3"
138
+ },
139
+ {
140
+ "path": "checkpoints/humanml3d_babel_fk_200k/assets/Mean.npy",
141
+ "size": 1180,
142
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
143
+ },
144
+ {
145
+ "path": "checkpoints/humanml3d_babel_fk_200k/assets/Std.npy",
146
+ "size": 1180,
147
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
148
+ },
149
+ {
150
+ "path": "checkpoints/humanml3d_babel_fk_200k/assets/W.npy",
151
+ "size": 553480,
152
+ "sha256": "778981f8025de72f2f98121ae018e81f69a8ce54fc0a2a8615c33706eb38d534"
153
+ }
154
+ ]
155
+ },
156
+ {
157
+ "name": "humanml3d_babel_path_200k",
158
+ "config": "configs/df_humanml3d_babel_263_path.yaml",
159
+ "step": 200000,
160
+ "files": [
161
+ {
162
+ "path": "checkpoints/humanml3d_babel_path_200k/model.ckpt",
163
+ "size": 2285618728,
164
+ "sha256": "d2e6a28baf6e1cfcd8c2061425ebdd82b1024091e9714729b087b428485cac0e"
165
+ },
166
+ {
167
+ "path": "checkpoints/humanml3d_babel_path_200k/config.yaml",
168
+ "size": 4534,
169
+ "sha256": "a86c8536002233d68512a644c0affc21274d5f9610aff58479118ec0538d8e28"
170
+ },
171
+ {
172
+ "path": "checkpoints/humanml3d_babel_path_200k/assets/Mean.npy",
173
+ "size": 1180,
174
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
175
+ },
176
+ {
177
+ "path": "checkpoints/humanml3d_babel_path_200k/assets/Std.npy",
178
+ "size": 1180,
179
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
180
+ },
181
+ {
182
+ "path": "checkpoints/humanml3d_babel_path_200k/assets/W.npy",
183
+ "size": 540928,
184
+ "sha256": "ebc7243c5185ab6203af8e0b912e05d6bd15919f9eda7391b548a870eefea373"
185
+ }
186
+ ]
187
+ },
188
+ {
189
+ "name": "seed_fk_300k",
190
+ "config": "configs/df_seed_138.yaml",
191
+ "step": 300000,
192
+ "files": [
193
+ {
194
+ "path": "checkpoints/seed_fk_300k/model.ckpt",
195
+ "size": 2280356842,
196
+ "sha256": "15caeff8faaa19f09437aee9f79088955708a32fd6ff609cc990338460dc130f"
197
+ },
198
+ {
199
+ "path": "checkpoints/seed_fk_300k/config.yaml",
200
+ "size": 3697,
201
+ "sha256": "41825321dce5aea7b3a384b4717b74aa7141c8e93988abc1987fd20bbdf6c686"
202
+ },
203
+ {
204
+ "path": "checkpoints/seed_fk_300k/assets/Mean.npy",
205
+ "size": 680,
206
+ "sha256": "98456f74f871856f48880c72d7340a776cb45dc88ba33e86491cfaf1f8bc9534"
207
+ },
208
+ {
209
+ "path": "checkpoints/seed_fk_300k/assets/Std.npy",
210
+ "size": 680,
211
+ "sha256": "5ea97221c9a6d471d12b92d6562d4214d1080fb398e96e43f628b393a70b2b96"
212
+ },
213
+ {
214
+ "path": "checkpoints/seed_fk_300k/assets/W.npy",
215
+ "size": 152480,
216
+ "sha256": "d5c81bdf26ff2347bb59840ab3bb52b179d9488fbb5ff88ea65f3e76fc9c5502"
217
+ }
218
+ ]
219
+ },
220
+ {
221
+ "name": "seed_path_fk_300k",
222
+ "config": "configs/df_seed_138_path.yaml",
223
+ "step": 300000,
224
+ "files": [
225
+ {
226
+ "path": "checkpoints/seed_path_fk_300k/model.ckpt",
227
+ "size": 2280293224,
228
+ "sha256": "d60a085a280962f66669fbeb12568e6f6cd1b67304647c1b73fc847862498ef6"
229
+ },
230
+ {
231
+ "path": "checkpoints/seed_path_fk_300k/config.yaml",
232
+ "size": 3811,
233
+ "sha256": "2c414967a1e1ad2d6e959e4a1ce7fc22d43086a69340d245f16a5e41bcef850e"
234
+ },
235
+ {
236
+ "path": "checkpoints/seed_path_fk_300k/assets/Mean.npy",
237
+ "size": 680,
238
+ "sha256": "98456f74f871856f48880c72d7340a776cb45dc88ba33e86491cfaf1f8bc9534"
239
+ },
240
+ {
241
+ "path": "checkpoints/seed_path_fk_300k/assets/Std.npy",
242
+ "size": 680,
243
+ "sha256": "5ea97221c9a6d471d12b92d6562d4214d1080fb398e96e43f628b393a70b2b96"
244
+ },
245
+ {
246
+ "path": "checkpoints/seed_path_fk_300k/assets/W.npy",
247
+ "size": 145928,
248
+ "sha256": "0effaf2383a21f4e5a1bf03aed5841060cd45226ac93177e8e09ecd9dbf540b5"
249
+ }
250
+ ]
251
+ }
252
+ ],
253
+ "dependencies": {
254
+ "repository": "AlayaLab/FloodDiffusion2",
255
+ "revision": "alaya-initial",
256
+ "archive": "deps.zip",
257
+ "size": 12992154753,
258
+ "sha256": "bd12f617474bf9f2a4321308b6ee3697b34b11a4ce77f47ff7d50db0d6fe7d85",
259
+ "files": [
260
+ {
261
+ "path": "deps/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth",
262
+ "size": 11361920418,
263
+ "crc32": "201263e2"
264
+ },
265
+ {
266
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/special_tokens_map.json",
267
+ "size": 6623,
268
+ "crc32": "52b4332b"
269
+ },
270
+ {
271
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/spiece.model",
272
+ "size": 4548313,
273
+ "crc32": "3d61aeda"
274
+ },
275
+ {
276
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/tokenizer.json",
277
+ "size": 16837417,
278
+ "crc32": "df89c52f"
279
+ },
280
+ {
281
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/tokenizer_config.json",
282
+ "size": 61728,
283
+ "crc32": "882574af"
284
+ },
285
+ {
286
+ "path": "deps/glove/our_vab_data.npy",
287
+ "size": 10077728,
288
+ "crc32": "16c5f298"
289
+ },
290
+ {
291
+ "path": "deps/glove/our_vab_idx.pkl",
292
+ "size": 79811,
293
+ "crc32": "d99790b0"
294
+ },
295
+ {
296
+ "path": "deps/glove/our_vab_words.pkl",
297
+ "size": 67470,
298
+ "crc32": "b473f73b"
299
+ },
300
+ {
301
+ "path": "deps/t2m/humanml3d/motion_encoder.pt",
302
+ "size": 62995445,
303
+ "crc32": "729e63ff"
304
+ },
305
+ {
306
+ "path": "deps/t2m/humanml3d/movement_encoder.pt",
307
+ "size": 7374081,
308
+ "crc32": "50493a9c"
309
+ },
310
+ {
311
+ "path": "deps/t2m/humanml3d/text_encoder.pt",
312
+ "size": 16406745,
313
+ "crc32": "b1579742"
314
+ },
315
+ {
316
+ "path": "deps/t2m/meta/mean.npy",
317
+ "size": 2232,
318
+ "crc32": "fe0c91c4"
319
+ },
320
+ {
321
+ "path": "deps/t2m/meta/std.npy",
322
+ "size": 2232,
323
+ "crc32": "8e2a77ab"
324
+ }
325
+ ]
326
+ }
327
+ }
checkpoints/humanml3d_babel_fk_200k/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0
3
+ size 1180
checkpoints/humanml3d_babel_fk_200k/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20
3
+ size 1180
checkpoints/humanml3d_babel_fk_200k/assets/W.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:778981f8025de72f2f98121ae018e81f69a8ce54fc0a2a8615c33706eb38d534
3
+ size 553480
checkpoints/humanml3d_babel_fk_200k/config.yaml ADDED
@@ -0,0 +1,165 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ model_dir: ${dirs.checkpoints}/humanml3d_babel_fk_200k
7
+ exp_name: df_humanml3d_babel_263
8
+ seed: 1234
9
+ debug: false
10
+ train: true
11
+ save_dir: ${dirs.outputs}
12
+ resume_ckpt: null
13
+ test_ckpt: ${dirs.checkpoints}/humanml3d_babel_fk_200k/model.ckpt
14
+ representation: humanml3d263
15
+ test_setting:
16
+ render: true
17
+ HumanML3D:
18
+ compare_folders:
19
+ - ${dirs.raw_data}/HumanML3D/HumanML3D263/renders
20
+ compare_names:
21
+ - Ground Truth
22
+ val_repeat: 1
23
+ logger:
24
+ wandb:
25
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
26
+ project: FloodDiffusion2
27
+ entity: ${oc.env:WANDB_ENTITY,null}
28
+ trainer:
29
+ max_steps: 300000
30
+ accelerator: gpu
31
+ devices: 1
32
+ log_every_n_steps: 100
33
+ precision: bf16-mixed
34
+ num_nodes: 1
35
+ validation:
36
+ validation_steps: 5000
37
+ test_steps: 5000
38
+ save_every_n_steps: 5000
39
+ save_top_k: 100
40
+ metrics:
41
+ t2m:
42
+ target: metrics.HumanML3D263.t2m.T2MMetrics
43
+ fid_target: original
44
+ params:
45
+ evaluate_text: true
46
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
47
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
48
+ wordvectorizer:
49
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
50
+ params:
51
+ meta_root: ${dirs.deps}/glove
52
+ prefix: our_vab
53
+ max_text_len: 20
54
+ textencoder:
55
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
56
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
57
+ params:
58
+ word_size: 300
59
+ pos_size: 15
60
+ hidden_size: 512
61
+ output_size: 512
62
+ moveencoder:
63
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
64
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
65
+ params:
66
+ input_size: 259
67
+ hidden_size: 512
68
+ output_size: 512
69
+ motionencoder:
70
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
71
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
72
+ params:
73
+ input_size: 512
74
+ hidden_size: 1024
75
+ output_size: 512
76
+ data:
77
+ target: datasets.multi.MultiDataset
78
+ collate_fn: datasets.multi.collate_fn
79
+ train_bs: 128
80
+ val_bs: 32
81
+ test_bs: 16
82
+ num_workers: 8
83
+ datasets:
84
+ - target: datasets.babel.BabelDataset
85
+ train_meta_paths:
86
+ - path: ${dirs.raw_data}/BABEL/HumanML3D263/train.txt
87
+ name: BABEL
88
+ val_meta_paths: []
89
+ test_meta_paths:
90
+ - path: ${dirs.raw_data}/BABEL/HumanML3D263/test_min.txt
91
+ name: BABEL
92
+ feature_path: new_joint_vecs
93
+ text_path: texts
94
+ random_length: 0
95
+ min_length: 40
96
+ max_length: 99999
97
+ window_length: 200
98
+ feature_fps: 20
99
+ token_fps: 5
100
+ - target: datasets.humanml3d.HumanML3DDataset
101
+ train_meta_paths:
102
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/train.txt
103
+ name: HumanML3D
104
+ val_meta_paths:
105
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test.txt
106
+ name: HumanML3D
107
+ test_meta_paths:
108
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test_min.txt
109
+ name: HumanML3D
110
+ feature_path: new_joint_vecs
111
+ text_path: texts
112
+ random_length: 0
113
+ min_length: 40
114
+ max_length: 200
115
+ stream_mode: true
116
+ feature_fps: 20
117
+ token_fps: 5
118
+ model:
119
+ target: models.diffusion_forcing_wan.DiffForcingWanModel
120
+ ema_decay: 0.99
121
+ params:
122
+ schedule_config:
123
+ noise_type: linear
124
+ chunk_size: 30
125
+ steps: 30
126
+ sigma_type: zero
127
+ sigma_scale: 1.0
128
+ random_epsilon: 0.05
129
+ train_n_windows: 4
130
+ text_config:
131
+ len: 512
132
+ dim: 4096
133
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
134
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
135
+ input_dim: 263
136
+ mean_path: ${dirs.checkpoints}/humanml3d_babel_fk_200k/assets/Mean.npy
137
+ std_path: ${dirs.checkpoints}/humanml3d_babel_fk_200k/assets/Std.npy
138
+ input_keys:
139
+ feature: feature
140
+ feature_length: feature_length
141
+ text: text
142
+ text_end: feature_text_end
143
+ cfg_config:
144
+ text_scale: 4.0
145
+ null_scale: -3.0
146
+ prediction_type: vel
147
+ attn_type: partial
148
+ loss_W: ${dirs.checkpoints}/humanml3d_babel_fk_200k/assets/W.npy
149
+ loss_w_coefficient: 1.0
150
+ fk_matrix:
151
+ recipe: hml263_trace
152
+ optimizer:
153
+ target: AdamW
154
+ params:
155
+ lr: 0.0002
156
+ betas:
157
+ - 0.9
158
+ - 0.99
159
+ weight_decay: 0.0
160
+ eps: 1.0e-08
161
+ lr_scheduler:
162
+ target: CosineAnnealingLR
163
+ params:
164
+ T_max: 300000
165
+ eta_min: 1.0e-06
checkpoints/humanml3d_babel_fk_200k/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:216a6168daca6552495ff8415a7cc31d896e9dfb116e4165e1b777ac609e6ae3
3
+ size 2285685418
checkpoints/humanml3d_babel_path_200k/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0
3
+ size 1180
checkpoints/humanml3d_babel_path_200k/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20
3
+ size 1180
checkpoints/humanml3d_babel_path_200k/assets/W.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ebc7243c5185ab6203af8e0b912e05d6bd15919f9eda7391b548a870eefea373
3
+ size 540928
checkpoints/humanml3d_babel_path_200k/config.yaml ADDED
@@ -0,0 +1,167 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ model_dir: ${dirs.checkpoints}/humanml3d_babel_path_200k
7
+ exp_name: df_humanml3d_babel_263_path
8
+ seed: 1234
9
+ debug: false
10
+ train: true
11
+ save_dir: ${dirs.outputs}
12
+ resume_ckpt: null
13
+ test_ckpt: ${dirs.checkpoints}/humanml3d_babel_path_200k/model.ckpt
14
+ representation: humanml3d263
15
+ test_setting:
16
+ render: true
17
+ HumanML3D:
18
+ compare_folders:
19
+ - ${dirs.raw_data}/HumanML3D/HumanML3D263/renders
20
+ compare_names:
21
+ - Ground Truth
22
+ val_repeat: 1
23
+ logger:
24
+ wandb:
25
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
26
+ project: FloodDiffusion2
27
+ entity: ${oc.env:WANDB_ENTITY,null}
28
+ trainer:
29
+ max_steps: 300000
30
+ accelerator: gpu
31
+ devices: 1
32
+ log_every_n_steps: 100
33
+ precision: bf16-mixed
34
+ num_nodes: 1
35
+ validation:
36
+ validation_steps: 5000
37
+ test_steps: 5000
38
+ save_every_n_steps: 5000
39
+ save_top_k: 100
40
+ metrics:
41
+ t2m:
42
+ target: metrics.HumanML3D263.t2m.T2MMetrics
43
+ fid_target: original
44
+ params:
45
+ evaluate_text: true
46
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
47
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
48
+ wordvectorizer:
49
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
50
+ params:
51
+ meta_root: ${dirs.deps}/glove
52
+ prefix: our_vab
53
+ max_text_len: 20
54
+ textencoder:
55
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
56
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
57
+ params:
58
+ word_size: 300
59
+ pos_size: 15
60
+ hidden_size: 512
61
+ output_size: 512
62
+ moveencoder:
63
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
64
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
65
+ params:
66
+ input_size: 259
67
+ hidden_size: 512
68
+ output_size: 512
69
+ motionencoder:
70
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
71
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
72
+ params:
73
+ input_size: 512
74
+ hidden_size: 1024
75
+ output_size: 512
76
+ data:
77
+ target: datasets.multi.MultiDataset
78
+ collate_fn: datasets.multi.collate_fn
79
+ train_bs: 128
80
+ val_bs: 32
81
+ test_bs: 16
82
+ num_workers: 8
83
+ datasets:
84
+ - target: datasets.babel.BabelDataset
85
+ train_meta_paths:
86
+ - path: ${dirs.raw_data}/BABEL/HumanML3D263/train.txt
87
+ name: BABEL
88
+ val_meta_paths: []
89
+ test_meta_paths:
90
+ - path: ${dirs.raw_data}/BABEL/HumanML3D263/test_min.txt
91
+ name: BABEL
92
+ feature_path: new_joint_vecs
93
+ text_path: texts
94
+ random_length: 0
95
+ min_length: 40
96
+ max_length: 99999
97
+ window_length: 200
98
+ feature_fps: 20
99
+ token_fps: 5
100
+ - target: datasets.humanml3d_position.HumanML3DPositionDataset
101
+ train_meta_paths:
102
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/train.txt
103
+ name: HumanML3D
104
+ val_meta_paths:
105
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test.txt
106
+ name: HumanML3D
107
+ test_meta_paths:
108
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test_min.txt
109
+ name: HumanML3D
110
+ feature_path: new_joint_vecs
111
+ text_path: texts
112
+ random_length: 0
113
+ min_length: 40
114
+ max_length: 200
115
+ stream_mode: true
116
+ feature_fps: 20
117
+ token_fps: 5
118
+ model:
119
+ target: models.diffusion_forcing_position_wan.DiffForcingPositionWanModel
120
+ ema_decay: 0.99
121
+ params:
122
+ schedule_config:
123
+ noise_type: linear
124
+ chunk_size: 30
125
+ steps: 30
126
+ sigma_type: zero
127
+ sigma_scale: 1.0
128
+ random_epsilon: 0.05
129
+ train_n_windows: 4
130
+ text_config:
131
+ len: 512
132
+ dim: 4096
133
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
134
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
135
+ input_dim: 263
136
+ mean_path: ${dirs.checkpoints}/humanml3d_babel_path_200k/assets/Mean.npy
137
+ std_path: ${dirs.checkpoints}/humanml3d_babel_path_200k/assets/Std.npy
138
+ input_keys:
139
+ feature: feature
140
+ feature_length: feature_length
141
+ text: text
142
+ text_end: feature_text_end
143
+ cfg_config:
144
+ text_scale: 4.0
145
+ null_scale: -3.0
146
+ prediction_type: vel
147
+ attn_type: partial
148
+ representation: humanml3d263
149
+ root_dim: 3
150
+ loss_W: ${dirs.checkpoints}/humanml3d_babel_path_200k/assets/W.npy
151
+ loss_w_coefficient: 1.0
152
+ fk_matrix:
153
+ recipe: hml263_path260
154
+ optimizer:
155
+ target: AdamW
156
+ params:
157
+ lr: 0.0002
158
+ betas:
159
+ - 0.9
160
+ - 0.99
161
+ weight_decay: 0.0
162
+ eps: 1.0e-08
163
+ lr_scheduler:
164
+ target: CosineAnnealingLR
165
+ params:
166
+ T_max: 300000
167
+ eta_min: 1.0e-06
checkpoints/humanml3d_babel_path_200k/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d2e6a28baf6e1cfcd8c2061425ebdd82b1024091e9714729b087b428485cac0e
3
+ size 2285618728
checkpoints/humanml3d_fk_60k/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0
3
+ size 1180
checkpoints/humanml3d_fk_60k/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20
3
+ size 1180
checkpoints/humanml3d_fk_60k/assets/W.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a8411b10882a74d09479e20e8ec10d0208473127f23e9da4d3192e2fea145d6
3
+ size 553480
checkpoints/humanml3d_fk_60k/config.yaml ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ exp_name: df_humanml3d_263
7
+ seed: 1234
8
+ debug: false
9
+ train: true
10
+ save_dir: ${dirs.outputs}
11
+ resume_ckpt: null
12
+ test_ckpt: ${dirs.checkpoints}/humanml3d_fk_60k/model.ckpt
13
+ representation: humanml3d263
14
+ test_setting:
15
+ render: true
16
+ HumanML3D:
17
+ compare_folders:
18
+ - ${dirs.raw_data}/HumanML3D/renders
19
+ compare_names:
20
+ - Ground Truth
21
+ val_repeat: 1
22
+ logger:
23
+ wandb:
24
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
25
+ project: FloodDiffusion2
26
+ entity: ${oc.env:WANDB_ENTITY,null}
27
+ trainer:
28
+ max_steps: 100000
29
+ accelerator: gpu
30
+ devices: 1
31
+ log_every_n_steps: 100
32
+ precision: bf16-mixed
33
+ num_nodes: 1
34
+ validation:
35
+ validation_steps: 5000
36
+ test_steps: 5000
37
+ save_every_n_steps: 5000
38
+ save_top_k: 100
39
+ metrics:
40
+ t2m:
41
+ target: metrics.HumanML3D263.t2m.T2MMetrics
42
+ fid_target: original
43
+ params:
44
+ evaluate_text: true
45
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
46
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
47
+ wordvectorizer:
48
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
49
+ params:
50
+ meta_root: ${dirs.deps}/glove
51
+ prefix: our_vab
52
+ max_text_len: 20
53
+ textencoder:
54
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
55
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
56
+ params:
57
+ word_size: 300
58
+ pos_size: 15
59
+ hidden_size: 512
60
+ output_size: 512
61
+ moveencoder:
62
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
63
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
64
+ params:
65
+ input_size: 259
66
+ hidden_size: 512
67
+ output_size: 512
68
+ motionencoder:
69
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
70
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
71
+ params:
72
+ input_size: 512
73
+ hidden_size: 1024
74
+ output_size: 512
75
+ data:
76
+ target: datasets.humanml3d.HumanML3DDataset
77
+ collate_fn: datasets.humanml3d.collate_fn
78
+ train_bs: 128
79
+ val_bs: 32
80
+ test_bs: 16
81
+ num_workers: 8
82
+ train_meta_paths:
83
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/train.txt
84
+ name: HumanML3D
85
+ val_meta_paths:
86
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test.txt
87
+ name: HumanML3D
88
+ test_meta_paths:
89
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test_min.txt
90
+ name: HumanML3D
91
+ feature_path: new_joint_vecs
92
+ text_path: texts
93
+ random_length: 0
94
+ min_length: 40
95
+ max_length: 200
96
+ model:
97
+ target: models.diffusion_forcing_wan.DiffForcingWanModel
98
+ ema_decay: 0.99
99
+ params:
100
+ schedule_config:
101
+ noise_type: linear
102
+ chunk_size: 30
103
+ steps: 30
104
+ sigma_type: zero
105
+ sigma_scale: 1.0
106
+ random_epsilon: 0.05
107
+ train_n_windows: 8
108
+ text_config:
109
+ len: 512
110
+ dim: 4096
111
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
112
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
113
+ input_dim: 263
114
+ loss_W: ${dirs.checkpoints}/humanml3d_fk_60k/assets/W.npy
115
+ loss_w_coefficient: 1.0
116
+ mean_path: ${dirs.checkpoints}/humanml3d_fk_60k/assets/Mean.npy
117
+ std_path: ${dirs.checkpoints}/humanml3d_fk_60k/assets/Std.npy
118
+ input_keys:
119
+ feature: feature
120
+ feature_length: feature_length
121
+ text: text
122
+ text_end: feature_text_end
123
+ cfg_config:
124
+ text_scale: 4.0
125
+ null_scale: -3.0
126
+ prediction_type: vel
127
+ attn_type: partial
128
+ loss_W_require_trace_normalized: false
129
+ fk_matrix:
130
+ recipe: hml263_pos64norm_rest1
131
+ optimizer:
132
+ target: AdamW
133
+ params:
134
+ lr: 0.0002
135
+ betas:
136
+ - 0.9
137
+ - 0.99
138
+ weight_decay: 0.0
139
+ eps: 1.0e-08
140
+ lr_scheduler:
141
+ target: CosineAnnealingLR
142
+ params:
143
+ T_max: 100000
144
+ eta_min: 1.0e-06
checkpoints/humanml3d_fk_60k/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d48332bcd88d06098627ed895e8d708702290d6ed518b20e5573b6b959725a63
3
+ size 2285681258
checkpoints/humanml3d_path_fk_55k/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0
3
+ size 1180
checkpoints/humanml3d_path_fk_55k/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20
3
+ size 1180
checkpoints/humanml3d_path_fk_55k/assets/W.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ebc7243c5185ab6203af8e0b912e05d6bd15919f9eda7391b548a870eefea373
3
+ size 540928
checkpoints/humanml3d_path_fk_55k/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ exp_name: df_humanml3d_263_path
7
+ seed: 1234
8
+ debug: false
9
+ train: true
10
+ save_dir: ${dirs.outputs}
11
+ resume_ckpt: null
12
+ test_ckpt: ${dirs.checkpoints}/humanml3d_path_fk_55k/model.ckpt
13
+ representation: humanml3d263
14
+ test_setting:
15
+ render: true
16
+ HumanML3D:
17
+ compare_folders:
18
+ - ${dirs.raw_data}/HumanML3D/renders
19
+ compare_names:
20
+ - Ground Truth
21
+ val_repeat: 1
22
+ logger:
23
+ wandb:
24
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
25
+ project: FloodDiffusion2
26
+ entity: ${oc.env:WANDB_ENTITY,null}
27
+ trainer:
28
+ max_steps: 100000
29
+ accelerator: gpu
30
+ devices: 1
31
+ log_every_n_steps: 100
32
+ precision: bf16-mixed
33
+ num_nodes: 1
34
+ validation:
35
+ validation_steps: 5000
36
+ test_steps: 5000
37
+ save_every_n_steps: 5000
38
+ save_top_k: 100
39
+ metrics:
40
+ t2m:
41
+ target: metrics.HumanML3D263.t2m.T2MMetrics
42
+ fid_target: original
43
+ params:
44
+ evaluate_text: true
45
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
46
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
47
+ wordvectorizer:
48
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
49
+ params:
50
+ meta_root: ${dirs.deps}/glove
51
+ prefix: our_vab
52
+ max_text_len: 20
53
+ textencoder:
54
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
55
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
56
+ params:
57
+ word_size: 300
58
+ pos_size: 15
59
+ hidden_size: 512
60
+ output_size: 512
61
+ moveencoder:
62
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
63
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
64
+ params:
65
+ input_size: 259
66
+ hidden_size: 512
67
+ output_size: 512
68
+ motionencoder:
69
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
70
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
71
+ params:
72
+ input_size: 512
73
+ hidden_size: 1024
74
+ output_size: 512
75
+ data:
76
+ target: datasets.humanml3d_position.HumanML3DPositionDataset
77
+ collate_fn: datasets.humanml3d_position.collate_fn
78
+ train_bs: 128
79
+ val_bs: 32
80
+ test_bs: 16
81
+ num_workers: 8
82
+ train_meta_paths:
83
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/train.txt
84
+ name: HumanML3D
85
+ val_meta_paths:
86
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test.txt
87
+ name: HumanML3D
88
+ test_meta_paths:
89
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test_min.txt
90
+ name: HumanML3D
91
+ feature_path: new_joint_vecs
92
+ text_path: texts
93
+ random_length: 0
94
+ min_length: 40
95
+ max_length: 200
96
+ model:
97
+ target: models.diffusion_forcing_position_wan.DiffForcingPositionWanModel
98
+ ema_decay: 0.99
99
+ params:
100
+ schedule_config:
101
+ noise_type: linear
102
+ chunk_size: 30
103
+ steps: 30
104
+ sigma_type: zero
105
+ sigma_scale: 1.0
106
+ random_epsilon: 0.05
107
+ train_n_windows: 8
108
+ text_config:
109
+ len: 512
110
+ dim: 4096
111
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
112
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
113
+ input_dim: 263
114
+ representation: humanml3d263
115
+ root_dim: 3
116
+ loss_W: ${dirs.checkpoints}/humanml3d_path_fk_55k/assets/W.npy
117
+ loss_w_coefficient: 1.0
118
+ mean_path: ${dirs.checkpoints}/humanml3d_path_fk_55k/assets/Mean.npy
119
+ std_path: ${dirs.checkpoints}/humanml3d_path_fk_55k/assets/Std.npy
120
+ input_keys:
121
+ feature: feature
122
+ feature_length: feature_length
123
+ text: text
124
+ text_end: feature_text_end
125
+ cfg_config:
126
+ text_scale: 3.0
127
+ null_scale: -2.0
128
+ prediction_type: vel
129
+ attn_type: partial
130
+ fk_matrix:
131
+ recipe: hml263_path260
132
+ optimizer:
133
+ target: AdamW
134
+ params:
135
+ lr: 0.0002
136
+ betas:
137
+ - 0.9
138
+ - 0.99
139
+ weight_decay: 0.0
140
+ eps: 1.0e-08
141
+ lr_scheduler:
142
+ target: CosineAnnealingLR
143
+ params:
144
+ T_max: 100000
145
+ eta_min: 1.0e-06
checkpoints/humanml3d_path_fk_55k/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ee1239ee5458781ab84d21329d190bfcb45a23d12bec17df48924e000672d104
3
+ size 2285614312
checkpoints/ldf_263/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c8b1c3183972f4ba2e542eb8bc26945f5d13782d04b56345d6ab5c71e714dff6
3
+ size 144
checkpoints/ldf_263/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a4234b3c2ca19566cd74b749e1c1ede5cbc2ab245cfcc487891b14c056a2357d
3
+ size 144
checkpoints/ldf_263/config.yaml ADDED
@@ -0,0 +1,174 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ exp_name: ldf_263
7
+ seed: 1234
8
+ debug: false
9
+ train: true
10
+ save_dir: ${dirs.outputs}
11
+ resume_ckpt: null
12
+ test_ckpt: ${dirs.checkpoints}/ldf_263/model.ckpt
13
+ test_vae_ckpt: ${dirs.checkpoints}/vae_263/model.ckpt
14
+ test_vae:
15
+ target: models.vae_wan.VAEWanModel
16
+ ema_decay: 0.99
17
+ params:
18
+ input_dim: 263
19
+ z_dim: 4
20
+ test_setting:
21
+ render: true
22
+ recover_dim: 263
23
+ BABEL:
24
+ compare_folders:
25
+ - ${dirs.raw_data}/BABEL/HumanML3D263/animations
26
+ compare_names:
27
+ - Ground Truth
28
+ HumanML3D:
29
+ compare_folders:
30
+ - ${dirs.raw_data}/HumanML3D/HumanML3D263/animations
31
+ compare_names:
32
+ - Ground Truth
33
+ val_repeat: 1
34
+ logger:
35
+ wandb:
36
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
37
+ project: FloodDiffusion2
38
+ entity: ${oc.env:WANDB_ENTITY,null}
39
+ trainer:
40
+ max_steps: 300000
41
+ accelerator: gpu
42
+ devices: 1
43
+ log_every_n_steps: 100
44
+ precision: bf16-mixed
45
+ num_nodes: 1
46
+ validation:
47
+ validation_steps: 5000
48
+ test_steps: 5000
49
+ save_every_n_steps: 5000
50
+ save_top_k: 100
51
+ metrics:
52
+ dim: 263
53
+ t2m:
54
+ target: metrics.HumanML3D263.t2m.T2MMetrics
55
+ fid_target: original
56
+ params:
57
+ evaluate_text: true
58
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
59
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
60
+ wordvectorizer:
61
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
62
+ params:
63
+ meta_root: ${dirs.deps}/glove
64
+ prefix: our_vab
65
+ max_text_len: 20
66
+ textencoder:
67
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
68
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
69
+ params:
70
+ word_size: 300
71
+ pos_size: 15
72
+ hidden_size: 512
73
+ output_size: 512
74
+ moveencoder:
75
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
76
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
77
+ params:
78
+ input_size: 259
79
+ hidden_size: 512
80
+ output_size: 512
81
+ motionencoder:
82
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
83
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
84
+ params:
85
+ input_size: 512
86
+ hidden_size: 1024
87
+ output_size: 512
88
+ data:
89
+ target: datasets.multi.MultiDataset
90
+ collate_fn: datasets.multi.collate_fn
91
+ train_bs: 32
92
+ val_bs: 16
93
+ test_bs: 16
94
+ num_workers: 8
95
+ datasets:
96
+ - target: datasets.babel.BabelDataset
97
+ train_meta_paths:
98
+ - path: ${dirs.raw_data}/BABEL/HumanML3D263/train.txt
99
+ name: BABEL
100
+ val_meta_paths: []
101
+ test_meta_paths:
102
+ - path: ${dirs.raw_data}/BABEL/HumanML3D263/test_min.txt
103
+ name: BABEL
104
+ feature_path: new_joint_vecs
105
+ token_path: TOKENS_20260113_233726_vae_wan_z4_2250000
106
+ text_path: texts
107
+ random_length: 0
108
+ min_length: 5
109
+ max_length: 99999
110
+ window_length: 200
111
+ feature_fps: 20
112
+ token_fps: 5
113
+ - target: datasets.humanml3d.HumanML3DDataset
114
+ train_meta_paths:
115
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/train.txt
116
+ name: HumanML3D
117
+ val_meta_paths:
118
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test.txt
119
+ name: HumanML3D
120
+ test_meta_paths:
121
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test_min.txt
122
+ name: HumanML3D
123
+ feature_path: new_joint_vecs
124
+ token_path: TOKENS_20260113_233726_vae_wan_z4_2250000
125
+ text_path: texts
126
+ random_length: 0
127
+ min_length: 40
128
+ max_length: 200
129
+ stream_mode: true
130
+ model:
131
+ target: models.diffusion_forcing_wan.DiffForcingWanModel
132
+ ema_decay: 0.99
133
+ params:
134
+ schedule_config:
135
+ noise_type: linear
136
+ chunk_size: 5
137
+ steps: 10
138
+ sigma_type: zero
139
+ sigma_scale: 1.0
140
+ random_epsilon: 0.05
141
+ train_n_windows: 4
142
+ text_config:
143
+ len: 512
144
+ dim: 4096
145
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
146
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
147
+ input_dim: 4
148
+ mean_path: ${dirs.checkpoints}/ldf_263/assets/Mean.npy
149
+ std_path: ${dirs.checkpoints}/ldf_263/assets/Std.npy
150
+ input_keys:
151
+ feature: token
152
+ feature_length: token_length
153
+ text: text
154
+ text_end: token_text_end
155
+ cfg_config:
156
+ text_scale: 6.0
157
+ null_scale: -5.0
158
+ prediction_type: vel
159
+ attn_type: partial
160
+ optimizer:
161
+ target: AdamW
162
+ params:
163
+ lr: 0.0002
164
+ betas:
165
+ - 0.9
166
+ - 0.99
167
+ weight_decay: 0.0
168
+ eps: 1.0e-08
169
+ lr_scheduler:
170
+ target: CosineAnnealingLR
171
+ params:
172
+ T_max: 1000
173
+ eta_min: 1.0e-06
174
+ representation: humanml3d263
checkpoints/ldf_263/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:94cc2ea9c0c29213a7f615e721c5fe5264e96d0a95f7f67adc0a5a5081f69437
3
+ size 2274793067
checkpoints/seed_fk_300k/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98456f74f871856f48880c72d7340a776cb45dc88ba33e86491cfaf1f8bc9534
3
+ size 680
checkpoints/seed_fk_300k/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ea97221c9a6d471d12b92d6562d4214d1080fb398e96e43f628b393a70b2b96
3
+ size 680
checkpoints/seed_fk_300k/assets/W.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d5c81bdf26ff2347bb59840ab3bb52b179d9488fbb5ff88ea65f3e76fc9c5502
3
+ size 152480
checkpoints/seed_fk_300k/config.yaml ADDED
@@ -0,0 +1,143 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ exp_name: df_seed_138
7
+ seed: 1234
8
+ debug: false
9
+ train: true
10
+ save_dir: ${dirs.outputs}
11
+ resume_ckpt: null
12
+ test_ckpt: ${dirs.checkpoints}/seed_fk_300k/model.ckpt
13
+ representation: mei138
14
+ test_setting:
15
+ render: true
16
+ SEED:
17
+ compare_folders:
18
+ - ${dirs.raw_data}/SEED/MEI138/new_joint_vecs_uni
19
+ compare_names:
20
+ - Ground Truth
21
+ val_repeat: 1
22
+ logger:
23
+ wandb:
24
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
25
+ project: FloodDiffusion2
26
+ entity: ${oc.env:WANDB_ENTITY,null}
27
+ trainer:
28
+ max_steps: 300000
29
+ accelerator: gpu
30
+ devices: 1
31
+ log_every_n_steps: 100
32
+ precision: bf16-mixed
33
+ num_nodes: 1
34
+ validation:
35
+ validation_steps: 5000
36
+ test_steps: 5000
37
+ save_every_n_steps: 5000
38
+ save_top_k: 100
39
+ metrics:
40
+ t2m:
41
+ target: metrics.MEI138.t2m.T2MMetrics
42
+ fid_target: original
43
+ params:
44
+ evaluate_text: true
45
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
46
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
47
+ wordvectorizer:
48
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
49
+ params:
50
+ meta_root: ${dirs.deps}/glove
51
+ prefix: our_vab
52
+ max_text_len: 20
53
+ textencoder:
54
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
55
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
56
+ params:
57
+ word_size: 300
58
+ pos_size: 15
59
+ hidden_size: 512
60
+ output_size: 512
61
+ moveencoder:
62
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
63
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
64
+ params:
65
+ input_size: 259
66
+ hidden_size: 512
67
+ output_size: 512
68
+ motionencoder:
69
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
70
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
71
+ params:
72
+ input_size: 512
73
+ hidden_size: 1024
74
+ output_size: 512
75
+ data:
76
+ target: datasets.humanml3d.HumanML3DDataset
77
+ collate_fn: datasets.humanml3d.collate_fn
78
+ train_bs: 128
79
+ val_bs: 32
80
+ test_bs: 16
81
+ num_workers: 8
82
+ train_meta_paths:
83
+ - path: ${dirs.raw_data}/SEED/MEI138/train.txt
84
+ name: SEED
85
+ val_meta_paths:
86
+ - path: ${dirs.raw_data}/SEED/MEI138/test_content.txt
87
+ name: SEED
88
+ test_meta_paths:
89
+ - path: ${dirs.raw_data}/SEED/MEI138/test_min.txt
90
+ name: SEED
91
+ feature_path: new_joint_vecs_uni
92
+ text_path: texts
93
+ random_length: 0
94
+ min_length: 60
95
+ max_length: 300
96
+ model:
97
+ target: models.diffusion_forcing_wan.DiffForcingWanModel
98
+ ema_decay: 0.99
99
+ params:
100
+ schedule_config:
101
+ noise_type: linear
102
+ chunk_size: 30
103
+ steps: 30
104
+ sigma_type: zero
105
+ sigma_scale: 1.0
106
+ random_epsilon: 0.05
107
+ train_n_windows: 8
108
+ text_config:
109
+ len: 512
110
+ dim: 4096
111
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
112
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
113
+ input_dim: 138
114
+ loss_W: ${dirs.checkpoints}/seed_fk_300k/assets/W.npy
115
+ loss_w_coefficient: 1.0
116
+ mean_path: ${dirs.checkpoints}/seed_fk_300k/assets/Mean.npy
117
+ std_path: ${dirs.checkpoints}/seed_fk_300k/assets/Std.npy
118
+ input_keys:
119
+ feature: feature
120
+ feature_length: feature_length
121
+ text: text
122
+ text_end: feature_text_end
123
+ cfg_config:
124
+ text_scale: 2.0
125
+ null_scale: -1.0
126
+ prediction_type: vel
127
+ attn_type: partial
128
+ fk_matrix:
129
+ recipe: seed138_mesh
130
+ optimizer:
131
+ target: AdamW
132
+ params:
133
+ lr: 0.0002
134
+ betas:
135
+ - 0.9
136
+ - 0.99
137
+ weight_decay: 0.0
138
+ eps: 1.0e-08
139
+ lr_scheduler:
140
+ target: CosineAnnealingLR
141
+ params:
142
+ T_max: 300000
143
+ eta_min: 1.0e-06
checkpoints/seed_fk_300k/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:15caeff8faaa19f09437aee9f79088955708a32fd6ff609cc990338460dc130f
3
+ size 2280356842
checkpoints/seed_path_fk_300k/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:98456f74f871856f48880c72d7340a776cb45dc88ba33e86491cfaf1f8bc9534
3
+ size 680
checkpoints/seed_path_fk_300k/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ea97221c9a6d471d12b92d6562d4214d1080fb398e96e43f628b393a70b2b96
3
+ size 680
checkpoints/seed_path_fk_300k/assets/W.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0effaf2383a21f4e5a1bf03aed5841060cd45226ac93177e8e09ecd9dbf540b5
3
+ size 145928
checkpoints/seed_path_fk_300k/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ exp_name: df_seed_138_path
7
+ seed: 1234
8
+ debug: false
9
+ train: true
10
+ save_dir: ${dirs.outputs}
11
+ resume_ckpt: null
12
+ test_ckpt: ${dirs.checkpoints}/seed_path_fk_300k/model.ckpt
13
+ representation: mei138
14
+ test_setting:
15
+ render: true
16
+ SEED:
17
+ compare_folders:
18
+ - ${dirs.raw_data}/SEED/MEI138/new_joint_vecs_uni
19
+ compare_names:
20
+ - Ground Truth
21
+ val_repeat: 1
22
+ logger:
23
+ wandb:
24
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
25
+ project: FloodDiffusion2
26
+ entity: ${oc.env:WANDB_ENTITY,null}
27
+ trainer:
28
+ max_steps: 300000
29
+ accelerator: gpu
30
+ devices: 1
31
+ log_every_n_steps: 100
32
+ precision: bf16-mixed
33
+ num_nodes: 1
34
+ validation:
35
+ validation_steps: 5000
36
+ test_steps: 5000
37
+ save_every_n_steps: 5000
38
+ save_top_k: 100
39
+ metrics:
40
+ t2m:
41
+ target: metrics.MEI138.t2m.T2MMetrics
42
+ fid_target: original
43
+ params:
44
+ evaluate_text: true
45
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
46
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
47
+ wordvectorizer:
48
+ target: metrics.HumanML3D263.word_vectorizer.WordVectorizer
49
+ params:
50
+ meta_root: ${dirs.deps}/glove
51
+ prefix: our_vab
52
+ max_text_len: 20
53
+ textencoder:
54
+ target: metrics.HumanML3D263.t2m_evaluator.TextEncoderBiGRUCo
55
+ ckpt: ${dirs.deps}/t2m/humanml3d/text_encoder.pt
56
+ params:
57
+ word_size: 300
58
+ pos_size: 15
59
+ hidden_size: 512
60
+ output_size: 512
61
+ moveencoder:
62
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
63
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
64
+ params:
65
+ input_size: 259
66
+ hidden_size: 512
67
+ output_size: 512
68
+ motionencoder:
69
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
70
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
71
+ params:
72
+ input_size: 512
73
+ hidden_size: 1024
74
+ output_size: 512
75
+ data:
76
+ target: datasets.humanml3d_position.HumanML3DPositionDataset
77
+ collate_fn: datasets.humanml3d_position.collate_fn
78
+ train_bs: 128
79
+ val_bs: 32
80
+ test_bs: 16
81
+ num_workers: 8
82
+ train_meta_paths:
83
+ - path: ${dirs.raw_data}/SEED/MEI138/train.txt
84
+ name: SEED
85
+ val_meta_paths:
86
+ - path: ${dirs.raw_data}/SEED/MEI138/test_content.txt
87
+ name: SEED
88
+ test_meta_paths:
89
+ - path: ${dirs.raw_data}/SEED/MEI138/test_min.txt
90
+ name: SEED
91
+ feature_path: new_joint_vecs_uni
92
+ text_path: texts
93
+ random_length: 0
94
+ min_length: 60
95
+ max_length: 300
96
+ model:
97
+ target: models.diffusion_forcing_position_wan.DiffForcingPositionWanModel
98
+ ema_decay: 0.99
99
+ params:
100
+ schedule_config:
101
+ noise_type: linear
102
+ chunk_size: 30
103
+ steps: 30
104
+ sigma_type: zero
105
+ sigma_scale: 1.0
106
+ random_epsilon: 0.05
107
+ train_n_windows: 8
108
+ text_config:
109
+ len: 512
110
+ dim: 4096
111
+ checkpoint_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth
112
+ tokenizer_path: ${dirs.deps}/t5_umt5-xxl-enc-bf16/google/umt5-xxl
113
+ input_dim: 138
114
+ representation: mei138
115
+ root_dim: 3
116
+ loss_W: ${dirs.checkpoints}/seed_path_fk_300k/assets/W.npy
117
+ loss_w_coefficient: 1.0
118
+ mean_path: ${dirs.checkpoints}/seed_path_fk_300k/assets/Mean.npy
119
+ std_path: ${dirs.checkpoints}/seed_path_fk_300k/assets/Std.npy
120
+ input_keys:
121
+ feature: feature
122
+ feature_length: feature_length
123
+ text: text
124
+ text_end: feature_text_end
125
+ cfg_config:
126
+ text_scale: 2.0
127
+ null_scale: -1.0
128
+ prediction_type: vel
129
+ attn_type: partial
130
+ fk_matrix:
131
+ recipe: seed138_path135
132
+ optimizer:
133
+ target: AdamW
134
+ params:
135
+ lr: 0.0002
136
+ betas:
137
+ - 0.9
138
+ - 0.99
139
+ weight_decay: 0.0
140
+ eps: 1.0e-08
141
+ lr_scheduler:
142
+ target: CosineAnnealingLR
143
+ params:
144
+ T_max: 300000
145
+ eta_min: 1.0e-06
checkpoints/seed_path_fk_300k/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d60a085a280962f66669fbeb12568e6f6cd1b67304647c1b73fc847862498ef6
3
+ size 2280293224
checkpoints/vae_263/assets/Mean.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0
3
+ size 1180
checkpoints/vae_263/assets/Std.npy ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20
3
+ size 1180
checkpoints/vae_263/config.yaml ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ dirs:
2
+ deps: ./deps
3
+ raw_data: ./data
4
+ outputs: ./outputs
5
+ checkpoints: ./checkpoints
6
+ exp_name: vae_263
7
+ seed: 1234
8
+ debug: false
9
+ train: true
10
+ save_dir: ${dirs.outputs}
11
+ resume_ckpt: null
12
+ test_ckpt: ${dirs.checkpoints}/vae_263/model.ckpt
13
+ representation: humanml3d263
14
+ test_setting:
15
+ render: true
16
+ HumanML3D:
17
+ compare_folders:
18
+ - ${dirs.raw_data}/HumanML3D/HumanML3D263/animations
19
+ compare_names:
20
+ - Ground Truth
21
+ val_repeat: 1
22
+ logger:
23
+ wandb:
24
+ wandb_key: ${oc.env:WANDB_API_KEY,null}
25
+ project: FloodDiffusion2
26
+ entity: ${oc.env:WANDB_ENTITY,null}
27
+ trainer:
28
+ max_steps: 3000000
29
+ accelerator: gpu
30
+ devices: 1
31
+ log_every_n_steps: 50
32
+ num_nodes: 1
33
+ precision: 32-true
34
+ validation:
35
+ validation_steps: 1000
36
+ test_steps: 10000
37
+ save_every_n_steps: 10000
38
+ save_top_k: 100
39
+ metrics:
40
+ mr:
41
+ target: metrics.mr.MRMetrics
42
+ t2m:
43
+ target: metrics.HumanML3D263.t2m.T2MMetrics
44
+ params:
45
+ evaluate_text: false
46
+ metric_mean_path: ${dirs.deps}/t2m/meta/mean.npy
47
+ metric_std_path: ${dirs.deps}/t2m/meta/std.npy
48
+ moveencoder:
49
+ target: metrics.HumanML3D263.t2m_evaluator.MovementConvEncoder
50
+ ckpt: ${dirs.deps}/t2m/humanml3d/movement_encoder.pt
51
+ params:
52
+ input_size: 259
53
+ hidden_size: 512
54
+ output_size: 512
55
+ motionencoder:
56
+ target: metrics.HumanML3D263.t2m_evaluator.MotionEncoderBiGRUCo
57
+ ckpt: ${dirs.deps}/t2m/humanml3d/motion_encoder.pt
58
+ params:
59
+ input_size: 512
60
+ hidden_size: 1024
61
+ output_size: 512
62
+ data:
63
+ target: datasets.humanml3d.HumanML3DDataset
64
+ collate_fn: datasets.humanml3d.collate_fn
65
+ train_bs: 256
66
+ val_bs: 64
67
+ test_bs: 16
68
+ num_workers: 8
69
+ train_meta_paths:
70
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/train.txt
71
+ name: HumanML3D
72
+ val_meta_paths:
73
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test.txt
74
+ name: HumanML3D
75
+ test_meta_paths:
76
+ - path: ${dirs.raw_data}/HumanML3D/HumanML3D263/test_min.txt
77
+ name: HumanML3D
78
+ feature_path: new_joint_vecs
79
+ token_path: null
80
+ text_path: texts
81
+ min_length: 40
82
+ max_length: 200
83
+ window_length: 189
84
+ model:
85
+ target: models.vae_wan.VAEWanModel
86
+ ema_decay: 0.99
87
+ params:
88
+ input_dim: 263
89
+ z_dim: 4
90
+ mean_path: ${dirs.checkpoints}/vae_263/assets/Mean.npy
91
+ std_path: ${dirs.checkpoints}/vae_263/assets/Std.npy
92
+ optimizer:
93
+ target: AdamW
94
+ params:
95
+ lr: 0.0002
96
+ betas:
97
+ - 0.9
98
+ - 0.99
99
+ weight_decay: 0.0
100
+ eps: 1.0e-08
101
+ lr_scheduler:
102
+ target: diffusers.optimization.get_constant_schedule_with_warmup
103
+ params:
104
+ num_warmup_steps: 1000
checkpoints/vae_263/model.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b80ce00b538b24b874dd15ac84fa4f34d9469fe7ac7f8d99e60bea276c3918d4
3
+ size 350279555
dependencies.json ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repository": "AlayaLab/FloodDiffusion2",
3
+ "revision": "alaya-initial",
4
+ "archive": "deps.zip",
5
+ "size": 12992154753,
6
+ "sha256": "bd12f617474bf9f2a4321308b6ee3697b34b11a4ce77f47ff7d50db0d6fe7d85",
7
+ "files": [
8
+ {
9
+ "path": "deps/t5_umt5-xxl-enc-bf16/models_t5_umt5-xxl-enc-bf16.pth",
10
+ "size": 11361920418,
11
+ "crc32": "201263e2"
12
+ },
13
+ {
14
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/special_tokens_map.json",
15
+ "size": 6623,
16
+ "crc32": "52b4332b"
17
+ },
18
+ {
19
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/spiece.model",
20
+ "size": 4548313,
21
+ "crc32": "3d61aeda"
22
+ },
23
+ {
24
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/tokenizer.json",
25
+ "size": 16837417,
26
+ "crc32": "df89c52f"
27
+ },
28
+ {
29
+ "path": "deps/t5_umt5-xxl-enc-bf16/google/umt5-xxl/tokenizer_config.json",
30
+ "size": 61728,
31
+ "crc32": "882574af"
32
+ },
33
+ {
34
+ "path": "deps/glove/our_vab_data.npy",
35
+ "size": 10077728,
36
+ "crc32": "16c5f298"
37
+ },
38
+ {
39
+ "path": "deps/glove/our_vab_idx.pkl",
40
+ "size": 79811,
41
+ "crc32": "d99790b0"
42
+ },
43
+ {
44
+ "path": "deps/glove/our_vab_words.pkl",
45
+ "size": 67470,
46
+ "crc32": "b473f73b"
47
+ },
48
+ {
49
+ "path": "deps/t2m/humanml3d/motion_encoder.pt",
50
+ "size": 62995445,
51
+ "crc32": "729e63ff"
52
+ },
53
+ {
54
+ "path": "deps/t2m/humanml3d/movement_encoder.pt",
55
+ "size": 7374081,
56
+ "crc32": "50493a9c"
57
+ },
58
+ {
59
+ "path": "deps/t2m/humanml3d/text_encoder.pt",
60
+ "size": 16406745,
61
+ "crc32": "b1579742"
62
+ },
63
+ {
64
+ "path": "deps/t2m/meta/mean.npy",
65
+ "size": 2232,
66
+ "crc32": "fe0c91c4"
67
+ },
68
+ {
69
+ "path": "deps/t2m/meta/std.npy",
70
+ "size": 2232,
71
+ "crc32": "8e2a77ab"
72
+ }
73
+ ]
74
+ }
deps.zip ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bd12f617474bf9f2a4321308b6ee3697b34b11a4ce77f47ff7d50db0d6fe7d85
3
+ size 12992154753
publication_manifest.json ADDED
@@ -0,0 +1,402 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format_version": 1,
3
+ "release": "FloodDiffusion2",
4
+ "checkpoints": [
5
+ {
6
+ "config": "vae_263.yaml",
7
+ "global_step": 2250000,
8
+ "checkpoint": {
9
+ "path": "checkpoints/vae_263/model.ckpt",
10
+ "bytes": 350279555,
11
+ "sha256": "b80ce00b538b24b874dd15ac84fa4f34d9469fe7ac7f8d99e60bea276c3918d4"
12
+ },
13
+ "configuration": {
14
+ "path": "checkpoints/vae_263/config.yaml",
15
+ "bytes": 2581,
16
+ "sha256": "7047f20640a97538d8629bf91cd12c667d7992740894e051501eb1badd0f098e"
17
+ },
18
+ "assets": [
19
+ {
20
+ "path": "checkpoints/vae_263/assets/Mean.npy",
21
+ "bytes": 1180,
22
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
23
+ },
24
+ {
25
+ "path": "checkpoints/vae_263/assets/Std.npy",
26
+ "bytes": 1180,
27
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
28
+ }
29
+ ],
30
+ "tensor_count_verified": 638,
31
+ "preserved_training_state": [
32
+ "state_dict",
33
+ "ema_state",
34
+ "optimizer_states",
35
+ "lr_schedulers",
36
+ "loops",
37
+ "epoch",
38
+ "global_step"
39
+ ],
40
+ "removed_metadata": [
41
+ "callbacks"
42
+ ],
43
+ "normalization_dimension": 263,
44
+ "loss_matrix_validation": null
45
+ },
46
+ {
47
+ "config": "ldf_263.yaml",
48
+ "global_step": 180000,
49
+ "checkpoint": {
50
+ "path": "checkpoints/ldf_263/model.ckpt",
51
+ "bytes": 2274793067,
52
+ "sha256": "94cc2ea9c0c29213a7f615e721c5fe5264e96d0a95f7f67adc0a5a5081f69437"
53
+ },
54
+ "configuration": {
55
+ "path": "checkpoints/ldf_263/config.yaml",
56
+ "bytes": 4564,
57
+ "sha256": "4a252ff70dbd9a1864a9f0afb7e71f0842440d1cc8bcdf360ab1769cbcf65f41"
58
+ },
59
+ "assets": [
60
+ {
61
+ "path": "checkpoints/ldf_263/assets/Mean.npy",
62
+ "bytes": 144,
63
+ "sha256": "c8b1c3183972f4ba2e542eb8bc26945f5d13782d04b56345d6ab5c71e714dff6"
64
+ },
65
+ {
66
+ "path": "checkpoints/ldf_263/assets/Std.npy",
67
+ "bytes": 144,
68
+ "sha256": "a4234b3c2ca19566cd74b749e1c1ede5cbc2ab245cfcc487891b14c056a2357d"
69
+ }
70
+ ],
71
+ "tensor_count_verified": 1388,
72
+ "preserved_training_state": [
73
+ "state_dict",
74
+ "ema_state",
75
+ "optimizer_states",
76
+ "lr_schedulers",
77
+ "loops",
78
+ "epoch",
79
+ "global_step"
80
+ ],
81
+ "removed_metadata": [
82
+ "callbacks"
83
+ ],
84
+ "normalization_dimension": 4,
85
+ "loss_matrix_validation": null
86
+ },
87
+ {
88
+ "config": "df_humanml3d_263.yaml",
89
+ "global_step": 60000,
90
+ "checkpoint": {
91
+ "path": "checkpoints/humanml3d_fk_60k/model.ckpt",
92
+ "bytes": 2285681258,
93
+ "sha256": "d48332bcd88d06098627ed895e8d708702290d6ed518b20e5573b6b959725a63"
94
+ },
95
+ "configuration": {
96
+ "path": "checkpoints/humanml3d_fk_60k/config.yaml",
97
+ "bytes": 3811,
98
+ "sha256": "983449db24beb87f9d63fbde21927e7119990bf10d6af6e32324f1a5526a7b8f"
99
+ },
100
+ "assets": [
101
+ {
102
+ "path": "checkpoints/humanml3d_fk_60k/assets/Mean.npy",
103
+ "bytes": 1180,
104
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
105
+ },
106
+ {
107
+ "path": "checkpoints/humanml3d_fk_60k/assets/Std.npy",
108
+ "bytes": 1180,
109
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
110
+ },
111
+ {
112
+ "path": "checkpoints/humanml3d_fk_60k/assets/W.npy",
113
+ "bytes": 553480,
114
+ "sha256": "4a8411b10882a74d09479e20e8ec10d0208473127f23e9da4d3192e2fea145d6"
115
+ }
116
+ ],
117
+ "tensor_count_verified": 1389,
118
+ "preserved_training_state": [
119
+ "state_dict",
120
+ "ema_state",
121
+ "optimizer_states",
122
+ "lr_schedulers",
123
+ "loops",
124
+ "epoch",
125
+ "global_step"
126
+ ],
127
+ "removed_metadata": [
128
+ "callbacks"
129
+ ],
130
+ "normalization_dimension": 263,
131
+ "loss_matrix_validation": {
132
+ "dimension": 263,
133
+ "coefficient": 1.0,
134
+ "raw_matrix_trace": 263.0000009536743,
135
+ "trace_normalization_required": false,
136
+ "loader_roundtrip_fp32_exact": true
137
+ }
138
+ },
139
+ {
140
+ "config": "df_humanml3d_263_path.yaml",
141
+ "global_step": 55000,
142
+ "checkpoint": {
143
+ "path": "checkpoints/humanml3d_path_fk_55k/model.ckpt",
144
+ "bytes": 2285614312,
145
+ "sha256": "ee1239ee5458781ab84d21329d190bfcb45a23d12bec17df48924e000672d104"
146
+ },
147
+ "configuration": {
148
+ "path": "checkpoints/humanml3d_path_fk_55k/config.yaml",
149
+ "bytes": 3877,
150
+ "sha256": "efcb66b0ec75c721b418887fa9824285dcc8934dc98550218348303bda273332"
151
+ },
152
+ "assets": [
153
+ {
154
+ "path": "checkpoints/humanml3d_path_fk_55k/assets/Mean.npy",
155
+ "bytes": 1180,
156
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
157
+ },
158
+ {
159
+ "path": "checkpoints/humanml3d_path_fk_55k/assets/Std.npy",
160
+ "bytes": 1180,
161
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
162
+ },
163
+ {
164
+ "path": "checkpoints/humanml3d_path_fk_55k/assets/W.npy",
165
+ "bytes": 540928,
166
+ "sha256": "ebc7243c5185ab6203af8e0b912e05d6bd15919f9eda7391b548a870eefea373"
167
+ }
168
+ ],
169
+ "tensor_count_verified": 1391,
170
+ "preserved_training_state": [
171
+ "state_dict",
172
+ "ema_state",
173
+ "optimizer_states",
174
+ "lr_schedulers",
175
+ "loops",
176
+ "epoch",
177
+ "global_step"
178
+ ],
179
+ "removed_metadata": [
180
+ "callbacks"
181
+ ],
182
+ "normalization_dimension": 263,
183
+ "loss_matrix_validation": {
184
+ "dimension": 260,
185
+ "coefficient": 1.0,
186
+ "raw_matrix_trace": 259.9999974966049,
187
+ "trace_normalization_required": true,
188
+ "loader_roundtrip_fp32_exact": true
189
+ }
190
+ },
191
+ {
192
+ "config": "df_humanml3d_babel_263.yaml",
193
+ "global_step": 200000,
194
+ "checkpoint": {
195
+ "path": "checkpoints/humanml3d_babel_fk_200k/model.ckpt",
196
+ "bytes": 2285685418,
197
+ "sha256": "216a6168daca6552495ff8415a7cc31d896e9dfb116e4165e1b777ac609e6ae3"
198
+ },
199
+ "configuration": {
200
+ "path": "checkpoints/humanml3d_babel_fk_200k/config.yaml",
201
+ "bytes": 4434,
202
+ "sha256": "69ee7e4642b5baa3bba4e9e89e14a69c020cb936075ced8683be98426488eac3"
203
+ },
204
+ "assets": [
205
+ {
206
+ "path": "checkpoints/humanml3d_babel_fk_200k/assets/Mean.npy",
207
+ "bytes": 1180,
208
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
209
+ },
210
+ {
211
+ "path": "checkpoints/humanml3d_babel_fk_200k/assets/Std.npy",
212
+ "bytes": 1180,
213
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
214
+ },
215
+ {
216
+ "path": "checkpoints/humanml3d_babel_fk_200k/assets/W.npy",
217
+ "bytes": 553480,
218
+ "sha256": "778981f8025de72f2f98121ae018e81f69a8ce54fc0a2a8615c33706eb38d534"
219
+ }
220
+ ],
221
+ "tensor_count_verified": 1389,
222
+ "preserved_training_state": [
223
+ "state_dict",
224
+ "ema_state",
225
+ "optimizer_states",
226
+ "lr_schedulers",
227
+ "loops",
228
+ "epoch",
229
+ "global_step"
230
+ ],
231
+ "removed_metadata": [
232
+ "callbacks"
233
+ ],
234
+ "normalization_dimension": 263,
235
+ "loss_matrix_validation": {
236
+ "dimension": 263,
237
+ "coefficient": 1.0,
238
+ "raw_matrix_trace": 263.0000002384186,
239
+ "trace_normalization_required": true,
240
+ "loader_roundtrip_fp32_exact": true
241
+ }
242
+ },
243
+ {
244
+ "config": "df_humanml3d_babel_263_path.yaml",
245
+ "global_step": 200000,
246
+ "checkpoint": {
247
+ "path": "checkpoints/humanml3d_babel_path_200k/model.ckpt",
248
+ "bytes": 2285618728,
249
+ "sha256": "d2e6a28baf6e1cfcd8c2061425ebdd82b1024091e9714729b087b428485cac0e"
250
+ },
251
+ "configuration": {
252
+ "path": "checkpoints/humanml3d_babel_path_200k/config.yaml",
253
+ "bytes": 4534,
254
+ "sha256": "a86c8536002233d68512a644c0affc21274d5f9610aff58479118ec0538d8e28"
255
+ },
256
+ "assets": [
257
+ {
258
+ "path": "checkpoints/humanml3d_babel_path_200k/assets/Mean.npy",
259
+ "bytes": 1180,
260
+ "sha256": "bef8fea46b1c9ed21c378a378c66c5f6b55f7ecf2fdfa911670a7d020e7fffb0"
261
+ },
262
+ {
263
+ "path": "checkpoints/humanml3d_babel_path_200k/assets/Std.npy",
264
+ "bytes": 1180,
265
+ "sha256": "041dee3a9e09c97e5f66e10fb4b63b9410fb57ecc609a556ec0b65d8def2fb20"
266
+ },
267
+ {
268
+ "path": "checkpoints/humanml3d_babel_path_200k/assets/W.npy",
269
+ "bytes": 540928,
270
+ "sha256": "ebc7243c5185ab6203af8e0b912e05d6bd15919f9eda7391b548a870eefea373"
271
+ }
272
+ ],
273
+ "tensor_count_verified": 1391,
274
+ "preserved_training_state": [
275
+ "state_dict",
276
+ "ema_state",
277
+ "optimizer_states",
278
+ "lr_schedulers",
279
+ "loops",
280
+ "epoch",
281
+ "global_step"
282
+ ],
283
+ "removed_metadata": [
284
+ "callbacks"
285
+ ],
286
+ "normalization_dimension": 263,
287
+ "loss_matrix_validation": {
288
+ "dimension": 260,
289
+ "coefficient": 1.0,
290
+ "raw_matrix_trace": 259.9999974966049,
291
+ "trace_normalization_required": true,
292
+ "loader_roundtrip_fp32_exact": true
293
+ }
294
+ },
295
+ {
296
+ "config": "df_seed_138.yaml",
297
+ "global_step": 300000,
298
+ "checkpoint": {
299
+ "path": "checkpoints/seed_fk_300k/model.ckpt",
300
+ "bytes": 2280356842,
301
+ "sha256": "15caeff8faaa19f09437aee9f79088955708a32fd6ff609cc990338460dc130f"
302
+ },
303
+ "configuration": {
304
+ "path": "checkpoints/seed_fk_300k/config.yaml",
305
+ "bytes": 3697,
306
+ "sha256": "41825321dce5aea7b3a384b4717b74aa7141c8e93988abc1987fd20bbdf6c686"
307
+ },
308
+ "assets": [
309
+ {
310
+ "path": "checkpoints/seed_fk_300k/assets/Mean.npy",
311
+ "bytes": 680,
312
+ "sha256": "98456f74f871856f48880c72d7340a776cb45dc88ba33e86491cfaf1f8bc9534"
313
+ },
314
+ {
315
+ "path": "checkpoints/seed_fk_300k/assets/Std.npy",
316
+ "bytes": 680,
317
+ "sha256": "5ea97221c9a6d471d12b92d6562d4214d1080fb398e96e43f628b393a70b2b96"
318
+ },
319
+ {
320
+ "path": "checkpoints/seed_fk_300k/assets/W.npy",
321
+ "bytes": 152480,
322
+ "sha256": "d5c81bdf26ff2347bb59840ab3bb52b179d9488fbb5ff88ea65f3e76fc9c5502"
323
+ }
324
+ ],
325
+ "tensor_count_verified": 1389,
326
+ "preserved_training_state": [
327
+ "state_dict",
328
+ "ema_state",
329
+ "optimizer_states",
330
+ "lr_schedulers",
331
+ "loops",
332
+ "epoch",
333
+ "global_step"
334
+ ],
335
+ "removed_metadata": [
336
+ "callbacks"
337
+ ],
338
+ "normalization_dimension": 138,
339
+ "loss_matrix_validation": {
340
+ "dimension": 138,
341
+ "coefficient": 1.0,
342
+ "raw_matrix_trace": 137.99999964237213,
343
+ "trace_normalization_required": true,
344
+ "loader_roundtrip_fp32_exact": true
345
+ }
346
+ },
347
+ {
348
+ "config": "df_seed_138_path.yaml",
349
+ "global_step": 300000,
350
+ "checkpoint": {
351
+ "path": "checkpoints/seed_path_fk_300k/model.ckpt",
352
+ "bytes": 2280293224,
353
+ "sha256": "d60a085a280962f66669fbeb12568e6f6cd1b67304647c1b73fc847862498ef6"
354
+ },
355
+ "configuration": {
356
+ "path": "checkpoints/seed_path_fk_300k/config.yaml",
357
+ "bytes": 3811,
358
+ "sha256": "2c414967a1e1ad2d6e959e4a1ce7fc22d43086a69340d245f16a5e41bcef850e"
359
+ },
360
+ "assets": [
361
+ {
362
+ "path": "checkpoints/seed_path_fk_300k/assets/Mean.npy",
363
+ "bytes": 680,
364
+ "sha256": "98456f74f871856f48880c72d7340a776cb45dc88ba33e86491cfaf1f8bc9534"
365
+ },
366
+ {
367
+ "path": "checkpoints/seed_path_fk_300k/assets/Std.npy",
368
+ "bytes": 680,
369
+ "sha256": "5ea97221c9a6d471d12b92d6562d4214d1080fb398e96e43f628b393a70b2b96"
370
+ },
371
+ {
372
+ "path": "checkpoints/seed_path_fk_300k/assets/W.npy",
373
+ "bytes": 145928,
374
+ "sha256": "0effaf2383a21f4e5a1bf03aed5841060cd45226ac93177e8e09ecd9dbf540b5"
375
+ }
376
+ ],
377
+ "tensor_count_verified": 1391,
378
+ "preserved_training_state": [
379
+ "state_dict",
380
+ "ema_state",
381
+ "optimizer_states",
382
+ "lr_schedulers",
383
+ "loops",
384
+ "epoch",
385
+ "global_step"
386
+ ],
387
+ "removed_metadata": [
388
+ "callbacks"
389
+ ],
390
+ "normalization_dimension": 138,
391
+ "loss_matrix_validation": {
392
+ "dimension": 135,
393
+ "coefficient": 1.0,
394
+ "raw_matrix_trace": 135.0000001192093,
395
+ "trace_normalization_required": true,
396
+ "loader_roundtrip_fp32_exact": true
397
+ }
398
+ }
399
+ ],
400
+ "total_checkpoint_bytes": 16328322404,
401
+ "complete": true
402
+ }