ymyy307 commited on
Commit
06a2f8e
·
verified ·
1 Parent(s): 3c11df8

Upload folder using huggingface_hub (part 5)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +2 -0
  2. examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit-2511.py +24 -0
  3. examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit.py +21 -0
  4. examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen-Poster.py +29 -0
  5. examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen.py +29 -0
  6. examples/qwen_image/model_training/validate_lora/Qwen-Image-In-Context-Control-Union.py +19 -0
  7. examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control-V2.py +37 -0
  8. examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control.py +25 -0
  9. examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered.py +27 -0
  10. examples/qwen_image/model_training/validate_lora/Qwen-Image.py +18 -0
  11. examples/qwen_video_edit/model_inference/Qwen-Video-Edit.py +27 -0
  12. examples/qwen_video_edit/model_inference_low_vram/Qwen-Video-Edit.py +40 -0
  13. examples/qwen_video_edit/model_training/full/Qwen-Video-Edit.sh +19 -0
  14. examples/qwen_video_edit/model_training/full/accelerate_config_zero3.yaml +23 -0
  15. examples/qwen_video_edit/model_training/lora/Qwen-Video-Edit.sh +22 -0
  16. examples/qwen_video_edit/model_training/train.py +175 -0
  17. examples/qwen_video_edit/model_training/validate_full/Qwen-Video-Edit.py +25 -0
  18. examples/qwen_video_edit/model_training/validate_lora/Qwen-Video-Edit.py +23 -0
  19. examples/stable_diffusion/model_inference/stable-diffusion-v1-5.py +25 -0
  20. examples/stable_diffusion/model_inference_low_vram/stable-diffusion-v1-5.py +36 -0
  21. examples/stable_diffusion/model_training/full/stable-diffusion-v1-5.sh +15 -0
  22. examples/stable_diffusion/model_training/lora/stable-diffusion-v1-5.sh +17 -0
  23. examples/stable_diffusion/model_training/special/split_training/stable-diffusion-v1-5.sh +38 -0
  24. examples/stable_diffusion/model_training/special/split_training/validate.py +26 -0
  25. examples/stable_diffusion/model_training/train.py +154 -0
  26. examples/stable_diffusion/model_training/validate_full/stable-diffusion-v1-5.py +27 -0
  27. examples/stable_diffusion/model_training/validate_lora/stable-diffusion-v1-5.py +26 -0
  28. examples/stable_diffusion_xl/model_inference/stable-diffusion-xl-base-1.0.py +26 -0
  29. examples/stable_diffusion_xl/model_inference_low_vram/stable-diffusion-xl-base-1.0.py +37 -0
  30. examples/stable_diffusion_xl/model_training/full/stable-diffusion-xl-base-1.0.sh +15 -0
  31. examples/stable_diffusion_xl/model_training/lora/stable-diffusion-xl-base-1.0.sh +18 -0
  32. examples/stable_diffusion_xl/model_training/special/split_training/stable-diffusion-xl-base-1.0.sh +40 -0
  33. examples/stable_diffusion_xl/model_training/special/split_training/validate.py +27 -0
  34. examples/stable_diffusion_xl/model_training/train.py +159 -0
  35. examples/stable_diffusion_xl/model_training/validate_full/stable-diffusion-xl-base-1.0.py +28 -0
  36. examples/stable_diffusion_xl/model_training/validate_lora/stable-diffusion-xl-base-1.0.py +27 -0
  37. examples/wanvideo/README.md +3 -0
  38. examples/wanvideo/acceleration/Wan2.2-Animate-2-14B-usp.py +117 -0
  39. examples/wanvideo/acceleration/unified_sequence_parallel.py +26 -0
  40. examples/wanvideo/model_inference/LongCat-Video.py +35 -0
  41. examples/wanvideo/model_inference/Video-As-Prompt-Wan2.1-14B.py +49 -0
  42. examples/wanvideo/model_inference/Wan-Dancer-14B-global.py +48 -0
  43. examples/wanvideo/model_inference/Wan-Dancer-14B-local.py +52 -0
  44. examples/wanvideo/model_inference/Wan2.1-1.3b-speedcontrol-v1.py +34 -0
  45. examples/wanvideo/model_inference/Wan2.1-FLF2V-14B-720P.py +36 -0
  46. examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-Control.py +34 -0
  47. examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-InP.py +36 -0
  48. examples/wanvideo/model_inference/Wan2.1-Fun-14B-Control.py +34 -0
  49. examples/wanvideo/model_inference/Wan2.1-Fun-14B-InP.py +36 -0
  50. examples/wanvideo/model_inference/Wan2.1-Fun-V1.1-1.3B-Control-Camera.py +44 -0
.gitattributes CHANGED
@@ -34,3 +34,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  .github/workflows/logo.gif filter=lfs diff=lfs merge=lfs -text
 
 
 
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  .github/workflows/logo.gif filter=lfs diff=lfs merge=lfs -text
37
+ models/train/MiniMax-H3-Ref2VA-CineDance-filtered-multishot-native-partial/wandb_log/wandb/run-20260916_162111-qrso8o89/run-qrso8o89.wandb filter=lfs diff=lfs merge=lfs -text
38
+ models/train/MiniMax-H3-Ref2VA-CineDance-multishot-full-v2/wandb_log/wandb/run-20260831_033451-73yd8zqv/run-73yd8zqv.wandb filter=lfs diff=lfs merge=lfs -text
examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit-2511.py ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
4
+
5
+ pipe = QwenImagePipeline.from_pretrained(
6
+ torch_dtype=torch.bfloat16,
7
+ device="cuda",
8
+ model_configs=[
9
+ ModelConfig(model_id="Qwen/Qwen-Image-Edit-2511", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
10
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=None,
14
+ processor_config=ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="processor/"),
15
+ )
16
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Edit-2511_lora/epoch-4.safetensors")
17
+
18
+ prompt = "Change the color of the dress in Figure 1 to the color shown in Figure 2."
19
+ images = [
20
+ Image.open("data/example_image_dataset/edit/image1.jpg").resize((1024, 1024)),
21
+ Image.open("data/example_image_dataset/edit/image_color.jpg").resize((1024, 1024)),
22
+ ]
23
+ image = pipe(prompt, edit_image=images, seed=123, num_inference_steps=40, height=1024, width=1024, zero_cond_t=True)
24
+ image.save("image.jpg")
examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
4
+
5
+ pipe = QwenImagePipeline.from_pretrained(
6
+ torch_dtype=torch.bfloat16,
7
+ device="cuda",
8
+ model_configs=[
9
+ ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
10
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=None,
14
+ processor_config=ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="processor/"),
15
+ )
16
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Edit_lora/epoch-4.safetensors")
17
+
18
+ prompt = "将裙子改为粉色"
19
+ image = Image.open("data/example_image_dataset/edit/image1.jpg").resize((1024, 1024))
20
+ image = pipe(prompt, edit_image=image, seed=0, num_inference_steps=40, height=1024, width=1024)
21
+ image.save(f"image.jpg")
examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen-Poster.py ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
2
+ import torch
3
+ from PIL import Image
4
+
5
+
6
+ pipe = QwenImagePipeline.from_pretrained(
7
+ torch_dtype=torch.bfloat16,
8
+ device="cuda",
9
+ model_configs=[
10
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
12
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
13
+ ],
14
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
15
+ )
16
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-EliGen-Poster_lora/epoch-4.safetensors")
17
+
18
+
19
+ entity_prompts = ["A beautiful girl", "sign 'Entity Control'", "shorts", "shirt"]
20
+ global_prompt = "A beautiful girl wearing shirt and shorts in the street, holding a sign 'Entity Control'"
21
+ masks = [Image.open(f"data/example_image_dataset/eligen/{i}.png").convert('RGB') for i in range(len(entity_prompts))]
22
+
23
+ image = pipe(global_prompt,
24
+ seed=0,
25
+ height=1024,
26
+ width=1024,
27
+ eligen_entity_prompts=entity_prompts,
28
+ eligen_entity_masks=masks)
29
+ image.save("Qwen-Image-EliGen-Poster.jpg")
examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen.py ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
2
+ import torch
3
+ from PIL import Image
4
+
5
+
6
+ pipe = QwenImagePipeline.from_pretrained(
7
+ torch_dtype=torch.bfloat16,
8
+ device="cuda",
9
+ model_configs=[
10
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
12
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
13
+ ],
14
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
15
+ )
16
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-EliGen_lora/epoch-4.safetensors")
17
+
18
+
19
+ entity_prompts = ["A beautiful girl", "sign 'Entity Control'", "shorts", "shirt"]
20
+ global_prompt = "A beautiful girl wearing shirt and shorts in the street, holding a sign 'Entity Control'"
21
+ masks = [Image.open(f"data/example_image_dataset/eligen/{i}.png").convert('RGB') for i in range(len(entity_prompts))]
22
+
23
+ image = pipe(global_prompt,
24
+ seed=0,
25
+ height=1024,
26
+ width=1024,
27
+ eligen_entity_prompts=entity_prompts,
28
+ eligen_entity_masks=masks)
29
+ image.save("Qwen-Image_EliGen.jpg")
examples/qwen_image/model_training/validate_lora/Qwen-Image-In-Context-Control-Union.py ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from PIL import Image
2
+ import torch
3
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
4
+
5
+ pipe = QwenImagePipeline.from_pretrained(
6
+ torch_dtype=torch.bfloat16,
7
+ device="cuda",
8
+ model_configs=[
9
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
10
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
14
+ )
15
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-In-Context-Control-Union_lora/epoch-4.safetensors")
16
+ image = Image.open("data/example_image_dataset/canny/image_1.jpg").resize((1024, 1024))
17
+ prompt = "Context_Control. a dog"
18
+ image = pipe(prompt=prompt, seed=0, context_image=image, height=1024, width=1024)
19
+ image.save("image_context.jpg")
examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control-V2.py ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
2
+ from modelscope import dataset_snapshot_download
3
+ from PIL import Image
4
+ import torch
5
+
6
+ pipe = QwenImagePipeline.from_pretrained(
7
+ torch_dtype=torch.bfloat16,
8
+ device="cuda",
9
+ model_configs=[
10
+ ModelConfig(model_id="DiffSynth-Studio/Qwen-Image-Layered-Control", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
12
+ ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
13
+ ],
14
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
15
+ )
16
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Layered-Control-V2_lora/epoch-4.safetensors")
17
+
18
+ prompt = "Text 'APRIL'"
19
+ input_image = Image.open("data/example_image_dataset/layer_v2/image_1.png").convert("RGBA").resize((1024, 1024))
20
+ image = pipe(
21
+ prompt, seed=0,
22
+ height=1024, width=1024,
23
+ layer_input_image=input_image, layer_num=0,
24
+ num_inference_steps=10, cfg_scale=4,
25
+ )
26
+ image[0].save("image_prompt.png")
27
+
28
+ mask_image = Image.open("data/example_image_dataset/layer_v2/mask_2.png").convert("RGBA").resize((1024, 1024))
29
+ input_image = Image.open("data/example_image_dataset/layer_v2/image_2.png").convert("RGBA").resize((1024, 1024))
30
+ image = pipe(
31
+ prompt, seed=0,
32
+ height=1024, width=1024,
33
+ layer_input_image=input_image, layer_num=0,
34
+ context_image=mask_image,
35
+ num_inference_steps=10, cfg_scale=1.0,
36
+ )
37
+ image[0].save("image_mask.png")
examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
2
+ from diffsynth import load_state_dict
3
+ from PIL import Image
4
+ import torch
5
+
6
+
7
+ pipe = QwenImagePipeline.from_pretrained(
8
+ torch_dtype=torch.bfloat16,
9
+ device="cuda",
10
+ model_configs=[
11
+ ModelConfig(model_id="DiffSynth-Studio/Qwen-Image-Layered-Control", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
12
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
13
+ ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
14
+ ],
15
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
16
+ )
17
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Layered-Control_lora/epoch-4.safetensors")
18
+ prompt = "Text 'HELLO' and 'Have a great day'"
19
+ input_image = Image.open("data/example_image_dataset/layer/image.png").convert("RGBA").resize((864, 480))
20
+ images = pipe(
21
+ prompt, seed=0,
22
+ height=480, width=864,
23
+ layer_input_image=input_image, layer_num=0,
24
+ )
25
+ images[0].save("image.png")
examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
2
+ from diffsynth import load_state_dict
3
+ from PIL import Image
4
+ import torch
5
+
6
+
7
+ pipe = QwenImagePipeline.from_pretrained(
8
+ torch_dtype=torch.bfloat16,
9
+ device="cuda",
10
+ model_configs=[
11
+ ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
12
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
13
+ ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
14
+ ],
15
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
16
+ )
17
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Layered_lora/epoch-4.safetensors")
18
+ prompt = "a poster"
19
+ input_image = Image.open("data/example_image_dataset/layer/image.png").convert("RGBA").resize((864, 480))
20
+ images = pipe(
21
+ prompt, seed=0,
22
+ height=480, width=864,
23
+ layer_input_image=input_image, layer_num=3,
24
+ )
25
+ for i, image in enumerate(images):
26
+ if i == 0: continue # The first image is the input image.
27
+ image.save(f"image_{i}.png")
examples/qwen_image/model_training/validate_lora/Qwen-Image.py ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
2
+ import torch
3
+
4
+
5
+ pipe = QwenImagePipeline.from_pretrained(
6
+ torch_dtype=torch.bfloat16,
7
+ device="cuda",
8
+ model_configs=[
9
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
10
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
11
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
14
+ )
15
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Image_lora/epoch-4.safetensors")
16
+ prompt = "a dog"
17
+ image = pipe(prompt, seed=0)
18
+ image.save("image.jpg")
examples/qwen_video_edit/model_inference/Qwen-Video-Edit.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from modelscope import dataset_snapshot_download
3
+ from diffsynth.core import ModelConfig
4
+ from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
5
+ from diffsynth.utils.data import VideoData, save_video
6
+
7
+ dataset_snapshot_download(
8
+ "DiffSynth-Studio/diffsynth_example_dataset",
9
+ local_dir="data/diffsynth_example_dataset",
10
+ allow_file_pattern="qwen_video_edit/Qwen-Video-Edit/*"
11
+ )
12
+
13
+ edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
14
+ prompts = [
15
+ "Transform the video into Japanese anime style",
16
+ ]
17
+ pipe = QwenVideoEditPipeline.from_pretrained(
18
+ torch_dtype=torch.bfloat16,
19
+ device="cuda",
20
+ model_configs=[
21
+ ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors"),
22
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
23
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
24
+ ],
25
+ )
26
+ video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
27
+ save_video(video, "video_Qwen-Video-Edit.mp4", fps=16)
examples/qwen_video_edit/model_inference_low_vram/Qwen-Video-Edit.py ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from modelscope import dataset_snapshot_download
3
+ from diffsynth.core import ModelConfig
4
+ from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
5
+ from diffsynth.utils.data import VideoData, save_video
6
+
7
+ vram_config = {
8
+ "offload_dtype": torch.bfloat16,
9
+ "offload_device": "cpu",
10
+ "onload_dtype": torch.bfloat16,
11
+ "onload_device": "cpu",
12
+ "preparing_dtype": torch.bfloat16,
13
+ "preparing_device": "cuda",
14
+ "computation_dtype": torch.bfloat16,
15
+ "computation_device": "cuda",
16
+ }
17
+
18
+ dataset_snapshot_download(
19
+ "DiffSynth-Studio/diffsynth_example_dataset",
20
+ local_dir="data/diffsynth_example_dataset",
21
+ allow_file_pattern="qwen_video_edit/Qwen-Video-Edit/*"
22
+ )
23
+
24
+ edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
25
+
26
+ prompts = [
27
+ "Transform the video into Japanese anime style",
28
+ ]
29
+ pipe = QwenVideoEditPipeline.from_pretrained(
30
+ torch_dtype=torch.bfloat16,
31
+ device="cuda",
32
+ model_configs=[
33
+ ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors", **vram_config),
34
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors", **vram_config),
35
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth", **vram_config),
36
+ ],
37
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
38
+ )
39
+ video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
40
+ save_video(video, "video_Qwen-Video-Edit.mp4", fps=16)
examples/qwen_video_edit/model_training/full/Qwen-Video-Edit.sh ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "qwen_video_edit/Qwen-Video-Edit/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ accelerate launch examples/qwen_video_edit/model_training/train.py \
4
+ --dataset_base_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit \
5
+ --dataset_metadata_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/metadata.json \
6
+ --data_file_keys "video,input_video" \
7
+ --height 640 \
8
+ --width 384 \
9
+ --num_frames 45 \
10
+ --dataset_repeat 50 \
11
+ --model_id_with_origin_paths "yunpeng1998/Qwen-Video-Edit:360P/step-30000.safetensors,Qwen/Qwen-Image:text_encoder/model*.safetensors,Wan-AI/Wan2.1-T2V-1.3B:Wan2.1_VAE.pth" \
12
+ --learning_rate 1e-5 \
13
+ --num_epochs 2 \
14
+ --remove_prefix_in_ckpt "pipe.dit." \
15
+ --output_path "./models/train/Qwen-Video-Edit_full" \
16
+ --trainable_models "dit" \
17
+ --use_gradient_checkpointing \
18
+ --zero_cond_t \
19
+ --find_unused_parameters
examples/qwen_video_edit/model_training/full/accelerate_config_zero3.yaml ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ compute_environment: LOCAL_MACHINE
2
+ debug: false
3
+ deepspeed_config:
4
+ gradient_accumulation_steps: 1
5
+ offload_optimizer_device: none
6
+ offload_param_device: none
7
+ zero3_init_flag: true
8
+ zero3_save_16bit_model: true
9
+ zero_stage: 3
10
+ distributed_type: DEEPSPEED
11
+ downcast_bf16: 'no'
12
+ enable_cpu_affinity: false
13
+ machine_rank: 0
14
+ main_training_function: main
15
+ mixed_precision: bf16
16
+ num_machines: 1
17
+ num_processes: 8
18
+ rdzv_backend: static
19
+ same_network: true
20
+ tpu_env: []
21
+ tpu_use_cluster: false
22
+ tpu_use_sudo: false
23
+ use_cpu: false
examples/qwen_video_edit/model_training/lora/Qwen-Video-Edit.sh ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "qwen_video_edit/Qwen-Video-Edit/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ accelerate launch examples/qwen_video_edit/model_training/train.py \
4
+ --dataset_base_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit \
5
+ --dataset_metadata_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/metadata.json \
6
+ --data_file_keys "video,input_video" \
7
+ --height 640 \
8
+ --width 384 \
9
+ --num_frames 45 \
10
+ --dataset_repeat 50 \
11
+ --model_id_with_origin_paths "yunpeng1998/Qwen-Video-Edit:360P/step-30000.safetensors,Qwen/Qwen-Image:text_encoder/model*.safetensors,Wan-AI/Wan2.1-T2V-1.3B:Wan2.1_VAE.pth" \
12
+ --learning_rate 1e-4 \
13
+ --num_epochs 5 \
14
+ --remove_prefix_in_ckpt "pipe.dit." \
15
+ --output_path "./models/train/Qwen-Video-Edit_lora" \
16
+ --lora_base_model "dit" \
17
+ --lora_target_modules "to_q,to_k,to_v,add_q_proj,add_k_proj,add_v_proj,to_out.0,to_add_out,img_mlp.net.2,img_mod.1,txt_mlp.net.2,txt_mod.1" \
18
+ --lora_rank 32 \
19
+ --use_gradient_checkpointing \
20
+ --zero_cond_t \
21
+ --dataset_num_workers 8 \
22
+ --find_unused_parameters
examples/qwen_video_edit/model_training/train.py ADDED
@@ -0,0 +1,175 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch, os, argparse, accelerate, warnings
2
+ from diffsynth.core import UnifiedDataset, ModelConfig
3
+ from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
4
+ from diffsynth.diffusion import *
5
+ from diffsynth.core.data.operators import *
6
+ os.environ["TOKENIZERS_PARALLELISM"] = "false"
7
+
8
+
9
+ class QwenVideoEditTrainingModule(DiffusionTrainingModule):
10
+ def __init__(
11
+ self,
12
+ model_paths=None, model_id_with_origin_paths=None,
13
+ tokenizer_path=None, processor_path=None,
14
+ trainable_models=None,
15
+ lora_base_model=None, lora_target_modules="", lora_rank=32, lora_checkpoint=None,
16
+ preset_lora_path=None, preset_lora_model=None,
17
+ use_gradient_checkpointing=True,
18
+ use_gradient_checkpointing_offload=False,
19
+ extra_inputs=None,
20
+ fp8_models=None,
21
+ offload_models=None,
22
+ quant_options=None,
23
+ resume_from_checkpoint=None, remove_prefix_in_ckpt=None,
24
+ device="cpu",
25
+ task="sft",
26
+ zero_cond_t=False,
27
+ max_timestep_boundary=1.0,
28
+ min_timestep_boundary=0.0,
29
+ ):
30
+ super().__init__()
31
+ # Load models
32
+ model_configs = self.parse_model_configs(model_paths, model_id_with_origin_paths, fp8_models=fp8_models, offload_models=offload_models, quant_options=quant_options, device=device)
33
+ tokenizer_config = ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/") if tokenizer_path is None else ModelConfig(tokenizer_path)
34
+ processor_config = ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="processor/") if processor_path is None else ModelConfig(processor_path)
35
+ self.pipe = QwenVideoEditPipeline.from_pretrained(torch_dtype=torch.bfloat16, device=device, model_configs=model_configs, tokenizer_config=tokenizer_config, processor_config=processor_config)
36
+ self.pipe = self.split_pipeline_units(task, self.pipe, trainable_models, lora_base_model)
37
+ self.resume_from_checkpoint(resume_from_checkpoint, remove_prefix_in_ckpt)
38
+
39
+ # Training mode
40
+ self.switch_pipe_to_training_mode(
41
+ self.pipe, trainable_models,
42
+ lora_base_model, lora_target_modules, lora_rank, lora_checkpoint,
43
+ preset_lora_path, preset_lora_model,
44
+ task=task,
45
+ )
46
+
47
+ # Store other configs
48
+ self.use_gradient_checkpointing = use_gradient_checkpointing
49
+ self.use_gradient_checkpointing_offload = use_gradient_checkpointing_offload
50
+ self.extra_inputs = extra_inputs.split(",") if extra_inputs is not None else []
51
+ self.fp8_models = fp8_models
52
+ self.task = task
53
+ self.zero_cond_t = zero_cond_t
54
+ self.max_timestep_boundary = max_timestep_boundary
55
+ self.min_timestep_boundary = min_timestep_boundary
56
+ self.task_to_loss = {
57
+ "sft:data_process": lambda pipe, *args: args,
58
+ "sft": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
59
+ "sft:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
60
+ }
61
+
62
+ def get_pipeline_inputs(self, data):
63
+ inputs_posi = {"prompt": data["prompt"]}
64
+ inputs_nega = {"negative_prompt": ""}
65
+ input_video = data["input_video"]
66
+ num_frames = len(input_video)
67
+ inputs_shared = {
68
+ "edit_video": input_video,
69
+ "input_video": data["video"],
70
+ "chunk_id": 0,
71
+ "num_frames": num_frames,
72
+ "height": input_video[0].size[1],
73
+ "width": input_video[0].size[0],
74
+ "tiled": False,
75
+ "tile_size": (30, 52),
76
+ "tile_stride": (15, 26),
77
+ "cfg_scale": 1,
78
+ "rand_device": self.pipe.device,
79
+ "use_gradient_checkpointing": self.use_gradient_checkpointing,
80
+ "use_gradient_checkpointing_offload": self.use_gradient_checkpointing_offload,
81
+ "zero_cond_t": self.zero_cond_t,
82
+ "max_timestep_boundary": self.max_timestep_boundary,
83
+ "min_timestep_boundary": self.min_timestep_boundary,
84
+ }
85
+ inputs_shared = self.parse_extra_inputs(data, self.extra_inputs, inputs_shared)
86
+ return inputs_shared, inputs_posi, inputs_nega
87
+
88
+ def forward(self, data, inputs=None):
89
+ if inputs is None: inputs = self.get_pipeline_inputs(data)
90
+ inputs = self.transfer_data_to_device(inputs, self.pipe.device, self.pipe.torch_dtype)
91
+ for unit in self.pipe.units:
92
+ inputs = self.pipe.unit_runner(unit, self.pipe, *inputs)
93
+ loss = self.task_to_loss[self.task](self.pipe, *inputs)
94
+ return loss
95
+
96
+
97
+ def qwen_video_edit_parser():
98
+ parser = argparse.ArgumentParser(description="Simple example of a training script.")
99
+ parser = add_general_config(parser)
100
+ parser = add_video_size_config(parser)
101
+ parser.add_argument("--tokenizer_path", type=str, default=None, help="Path to tokenizer.")
102
+ parser.add_argument("--processor_path", type=str, default=None, help="Path to the processor. If provided, the processor will be used for image editing.")
103
+ parser.add_argument("--zero_cond_t", default=False, action="store_true", help="A special parameter introduced by Qwen-Image-Edit-2511. Please enable it for this model.")
104
+ parser.add_argument("--max_timestep_boundary", type=float, default=1.0, help="Max timestep boundary.")
105
+ parser.add_argument("--min_timestep_boundary", type=float, default=0.0, help="Min timestep boundary.")
106
+ parser.add_argument("--initialize_model_on_cpu", default=False, action="store_true", help="Whether to initialize models on CPU.")
107
+ return parser
108
+
109
+
110
+ if __name__ == "__main__":
111
+ parser = qwen_video_edit_parser()
112
+ args = parser.parse_args()
113
+ accelerator = accelerate.Accelerator(
114
+ gradient_accumulation_steps=args.gradient_accumulation_steps,
115
+ kwargs_handlers=[accelerate.DistributedDataParallelKwargs(find_unused_parameters=args.find_unused_parameters)],
116
+ )
117
+ dataset = UnifiedDataset(
118
+ base_path=args.dataset_base_path,
119
+ metadata_path=args.dataset_metadata_path,
120
+ repeat=args.dataset_repeat,
121
+ data_file_keys=args.data_file_keys.split(","),
122
+ main_data_operator=UnifiedDataset.default_video_operator(
123
+ base_path=args.dataset_base_path,
124
+ max_pixels=args.max_pixels,
125
+ height=args.height,
126
+ width=args.width,
127
+ height_division_factor=16,
128
+ width_division_factor=16,
129
+ num_frames=args.num_frames,
130
+ time_division_factor=4,
131
+ time_division_remainder=1,
132
+ ),
133
+ )
134
+ model = QwenVideoEditTrainingModule(
135
+ model_paths=args.model_paths,
136
+ model_id_with_origin_paths=args.model_id_with_origin_paths,
137
+ tokenizer_path=args.tokenizer_path,
138
+ processor_path=args.processor_path,
139
+ trainable_models=args.trainable_models,
140
+ lora_base_model=args.lora_base_model,
141
+ lora_target_modules=args.lora_target_modules,
142
+ lora_rank=args.lora_rank,
143
+ lora_checkpoint=args.lora_checkpoint,
144
+ preset_lora_path=args.preset_lora_path,
145
+ preset_lora_model=args.preset_lora_model,
146
+ use_gradient_checkpointing=args.use_gradient_checkpointing,
147
+ use_gradient_checkpointing_offload=args.use_gradient_checkpointing_offload,
148
+ extra_inputs=args.extra_inputs,
149
+ fp8_models=args.fp8_models,
150
+ offload_models=args.offload_models,
151
+ quant_options=args.quant_options,
152
+ resume_from_checkpoint=args.resume_from_checkpoint,
153
+ remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
154
+ task=args.task,
155
+ device="cpu" if (args.initialize_model_on_cpu or args.enable_model_cpu_offload) else accelerator.device,
156
+ zero_cond_t=args.zero_cond_t,
157
+ max_timestep_boundary=args.max_timestep_boundary,
158
+ min_timestep_boundary=args.min_timestep_boundary,
159
+ )
160
+ model_logger = ModelLogger(
161
+ args.output_path,
162
+ remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
163
+ enable_tensorboard_log=args.enable_tensorboard_log,
164
+ enable_swanlab_log=args.enable_swanlab_log,
165
+ swanlab_project=args.swanlab_project,
166
+ enable_wandb_log=args.enable_wandb_log,
167
+ wandb_project=args.wandb_project,
168
+ enable_csv_log=args.enable_csv_log,
169
+ )
170
+ launcher_map = {
171
+ "sft:data_process": launch_data_process_task,
172
+ "sft": launch_training_task,
173
+ "sft:train": launch_training_task,
174
+ }
175
+ launcher_map[args.task](accelerator, dataset, model, model_logger, args=args)
examples/qwen_video_edit/model_training/validate_full/Qwen-Video-Edit.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+
3
+ from diffsynth import load_state_dict
4
+ from diffsynth.core import ModelConfig
5
+ from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
6
+ from diffsynth.utils.data import VideoData, save_video
7
+
8
+ edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
9
+ prompts = [
10
+ "Transform the video into Japanese anime style",
11
+ ]
12
+ pipe = QwenVideoEditPipeline.from_pretrained(
13
+ torch_dtype=torch.bfloat16,
14
+ device="cuda",
15
+ model_configs=[
16
+ ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors"),
17
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
18
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
19
+ ],
20
+ )
21
+ state_dict = load_state_dict("models/train/Qwen-Video-Edit_full/epoch-1.safetensors")
22
+ pipe.dit.load_state_dict(state_dict)
23
+
24
+ video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
25
+ save_video(video, "video_Qwen-Video-Edit-full.mp4", fps=16)
examples/qwen_video_edit/model_training/validate_lora/Qwen-Video-Edit.py ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+
3
+ from diffsynth.core import ModelConfig
4
+ from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
5
+ from diffsynth.utils.data import VideoData, save_video
6
+
7
+ edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
8
+ prompts = [
9
+ "Transform the video into Japanese anime style",
10
+ ]
11
+ pipe = QwenVideoEditPipeline.from_pretrained(
12
+ torch_dtype=torch.bfloat16,
13
+ device="cuda",
14
+ model_configs=[
15
+ ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors"),
16
+ ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
17
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
18
+ ],
19
+ )
20
+ pipe.load_lora(pipe.dit, "models/train/Qwen-Video-Edit_lora/epoch-4.safetensors")
21
+
22
+ video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
23
+ save_video(video, "video_Qwen-Video-Edit-lora.mp4", fps=16)
examples/stable_diffusion/model_inference/stable-diffusion-v1-5.py ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
4
+
5
+ pipe = StableDiffusionPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
10
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
11
+ ],
12
+ tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
13
+ )
14
+
15
+ image = pipe(
16
+ prompt="a photo of an astronaut riding a horse on mars, high quality, detailed",
17
+ negative_prompt="blurry, low quality, deformed",
18
+ cfg_scale=7.5,
19
+ height=512,
20
+ width=512,
21
+ seed=42,
22
+ rand_device="cuda",
23
+ num_inference_steps=50,
24
+ )
25
+ image.save("image.jpg")
examples/stable_diffusion/model_inference_low_vram/stable-diffusion-v1-5.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
4
+
5
+ vram_config = {
6
+ "offload_dtype": torch.float32,
7
+ "offload_device": "cpu",
8
+ "onload_dtype": torch.float32,
9
+ "onload_device": "cpu",
10
+ "preparing_dtype": torch.float32,
11
+ "preparing_device": "cuda",
12
+ "computation_dtype": torch.float32,
13
+ "computation_device": "cuda",
14
+ }
15
+ pipe = StableDiffusionPipeline.from_pretrained(
16
+ torch_dtype=torch.float32,
17
+ model_configs=[
18
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors", **vram_config),
19
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors", **vram_config),
20
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config),
21
+ ],
22
+ tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
23
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
24
+ )
25
+
26
+ image = pipe(
27
+ prompt="a photo of an astronaut riding a horse on mars, high quality, detailed",
28
+ negative_prompt="blurry, low quality, deformed",
29
+ cfg_scale=7.5,
30
+ height=512,
31
+ width=512,
32
+ seed=42,
33
+ rand_device="cuda",
34
+ num_inference_steps=50,
35
+ )
36
+ image.save("image.jpg")
examples/stable_diffusion/model_training/full/stable-diffusion-v1-5.sh ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion/stable-diffusion-v1-5/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ accelerate launch examples/stable_diffusion/model_training/train.py \
4
+ --dataset_base_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5 \
5
+ --dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5/metadata.csv \
6
+ --height 512 \
7
+ --width 512 \
8
+ --dataset_repeat 50 \
9
+ --model_id_with_origin_paths "AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors" \
10
+ --learning_rate 1e-5 \
11
+ --num_epochs 2 \
12
+ --trainable_models "unet" \
13
+ --remove_prefix_in_ckpt "pipe.unet." \
14
+ --output_path "./models/train/stable-diffusion-v1-5_full" \
15
+ --use_gradient_checkpointing
examples/stable_diffusion/model_training/lora/stable-diffusion-v1-5.sh ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion/stable-diffusion-v1-5/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ accelerate launch examples/stable_diffusion/model_training/train.py \
4
+ --dataset_base_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5 \
5
+ --dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5/metadata.csv \
6
+ --height 512 \
7
+ --width 512 \
8
+ --dataset_repeat 50 \
9
+ --model_id_with_origin_paths "AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors" \
10
+ --learning_rate 1e-4 \
11
+ --num_epochs 5 \
12
+ --remove_prefix_in_ckpt "pipe.unet." \
13
+ --output_path "./models/train/stable-diffusion-v1-5_lora" \
14
+ --lora_base_model "unet" \
15
+ --lora_target_modules "" \
16
+ --lora_rank 32 \
17
+ --use_gradient_checkpointing
examples/stable_diffusion/model_training/special/split_training/stable-diffusion-v1-5.sh ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion/stable-diffusion-v1-5/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ # Stage 1: cache deterministic preprocessing outputs.
4
+ accelerate launch examples/stable_diffusion/model_training/train.py \
5
+ --dataset_base_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5 \
6
+ --dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5/metadata.csv \
7
+ --height 512 \
8
+ --width 512 \
9
+ --dataset_repeat 1 \
10
+ --model_id_with_origin_paths AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors \
11
+ --learning_rate 1e-4 \
12
+ --num_epochs 5 \
13
+ --remove_prefix_in_ckpt pipe.unet. \
14
+ --output_path ./models/train/stable-diffusion-v1-5_split_cache \
15
+ --lora_base_model unet \
16
+ --lora_target_modules '' \
17
+ --lora_rank 32 \
18
+ --use_gradient_checkpointing \
19
+ --offload_models AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors \
20
+ --task sft:data_process
21
+
22
+ # Stage 2: train LoRA from the cached dataset.
23
+ accelerate launch examples/stable_diffusion/model_training/train.py \
24
+ --dataset_base_path ./models/train/stable-diffusion-v1-5_split_cache \
25
+ --height 512 \
26
+ --width 512 \
27
+ --dataset_repeat 50 \
28
+ --model_id_with_origin_paths AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors \
29
+ --learning_rate 1e-4 \
30
+ --num_epochs 5 \
31
+ --remove_prefix_in_ckpt pipe.unet. \
32
+ --output_path ./models/train/stable-diffusion-v1-5_split \
33
+ --lora_base_model unet \
34
+ --lora_target_modules '' \
35
+ --lora_rank 32 \
36
+ --use_gradient_checkpointing \
37
+ --offload_models AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors \
38
+ --task sft:train
examples/stable_diffusion/model_training/special/split_training/validate.py ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
4
+
5
+ pipe = StableDiffusionPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
10
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
11
+ ],
12
+ tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
13
+ )
14
+ pipe.load_lora(pipe.unet, './models/train/stable-diffusion-v1-5_split/epoch-4.safetensors')
15
+
16
+ image = pipe(
17
+ prompt="a dog",
18
+ negative_prompt="blurry, low quality, deformed",
19
+ cfg_scale=7.5,
20
+ height=512,
21
+ width=512,
22
+ seed=42,
23
+ rand_device="cuda",
24
+ num_inference_steps=50,
25
+ )
26
+ image.save('split_training_stable-diffusion-v1-5.jpg')
examples/stable_diffusion/model_training/train.py ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch, os, argparse, accelerate
2
+ from diffsynth.core import UnifiedDataset
3
+ from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline, ModelConfig
4
+ from diffsynth.diffusion import *
5
+ os.environ["TOKENIZERS_PARALLELISM"] = "false"
6
+
7
+
8
+ class StableDiffusionTrainingModule(DiffusionTrainingModule):
9
+ def __init__(
10
+ self,
11
+ model_paths=None, model_id_with_origin_paths=None,
12
+ tokenizer_path=None,
13
+ trainable_models=None,
14
+ lora_base_model=None, lora_target_modules="", lora_rank=32, lora_checkpoint=None,
15
+ preset_lora_path=None, preset_lora_model=None,
16
+ use_gradient_checkpointing=True,
17
+ use_gradient_checkpointing_offload=False,
18
+ extra_inputs=None,
19
+ fp8_models=None,
20
+ offload_models=None,
21
+ quant_options=None,
22
+ resume_from_checkpoint=None, remove_prefix_in_ckpt=None,
23
+ device="cpu",
24
+ task="sft",
25
+ ):
26
+ super().__init__()
27
+ # Load models
28
+ model_configs = self.parse_model_configs(model_paths, model_id_with_origin_paths, fp8_models=fp8_models, offload_models=offload_models, quant_options=quant_options, device=device)
29
+ tokenizer_config = self.parse_path_or_model_id(tokenizer_path, ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"))
30
+ self.pipe = StableDiffusionPipeline.from_pretrained(torch_dtype=torch.float32, device=device, model_configs=model_configs, tokenizer_config=tokenizer_config)
31
+ self.pipe = self.split_pipeline_units(task, self.pipe, trainable_models, lora_base_model)
32
+ self.resume_from_checkpoint(resume_from_checkpoint, remove_prefix_in_ckpt)
33
+
34
+ # Training mode
35
+ self.switch_pipe_to_training_mode(
36
+ self.pipe, trainable_models,
37
+ lora_base_model, lora_target_modules, lora_rank, lora_checkpoint,
38
+ preset_lora_path, preset_lora_model,
39
+ task=task,
40
+ )
41
+
42
+ # Other configs
43
+ self.use_gradient_checkpointing = use_gradient_checkpointing
44
+ self.use_gradient_checkpointing_offload = use_gradient_checkpointing_offload
45
+ self.extra_inputs = extra_inputs.split(",") if extra_inputs is not None else []
46
+ self.fp8_models = fp8_models
47
+ self.task = task
48
+ self.task_to_loss = {
49
+ "sft:data_process": lambda pipe, *args: args,
50
+ "direct_distill:data_process": lambda pipe, *args: args,
51
+ "sft": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
52
+ "sft:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
53
+ "direct_distill": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
54
+ "direct_distill:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
55
+ }
56
+
57
+ def get_pipeline_inputs(self, data):
58
+ inputs_posi = {"prompt": data["prompt"]}
59
+ inputs_nega = {"negative_prompt": ""}
60
+ inputs_shared = {
61
+ # Assume you are using this pipeline for inference,
62
+ # please fill in the input parameters.
63
+ "input_image": data["image"],
64
+ "height": data["image"].size[1],
65
+ "width": data["image"].size[0],
66
+ # Please do not modify the following parameters
67
+ # unless you clearly know what this will cause.
68
+ "cfg_scale": 1,
69
+ "rand_device": self.pipe.device,
70
+ "use_gradient_checkpointing": self.use_gradient_checkpointing,
71
+ "use_gradient_checkpointing_offload": self.use_gradient_checkpointing_offload,
72
+ }
73
+ inputs_shared = self.parse_extra_inputs(data, self.extra_inputs, inputs_shared)
74
+ return inputs_shared, inputs_posi, inputs_nega
75
+
76
+ def forward(self, data, inputs=None):
77
+ if inputs is None: inputs = self.get_pipeline_inputs(data)
78
+ inputs = self.transfer_data_to_device(inputs, self.pipe.device, self.pipe.torch_dtype)
79
+ for unit in self.pipe.units:
80
+ inputs = self.pipe.unit_runner(unit, self.pipe, *inputs)
81
+ loss = self.task_to_loss[self.task](self.pipe, *inputs)
82
+ return loss
83
+
84
+
85
+ def parser():
86
+ parser = argparse.ArgumentParser(description="Simple example of a training script.")
87
+ parser = add_general_config(parser)
88
+ parser = add_image_size_config(parser)
89
+ parser.add_argument("--tokenizer_path", type=str, default=None, help="Path to tokenizer.")
90
+ return parser
91
+
92
+
93
+ if __name__ == "__main__":
94
+ parser = parser()
95
+ args = parser.parse_args()
96
+ accelerator = accelerate.Accelerator(
97
+ gradient_accumulation_steps=args.gradient_accumulation_steps,
98
+ kwargs_handlers=[accelerate.DistributedDataParallelKwargs(find_unused_parameters=args.find_unused_parameters)],
99
+ )
100
+ dataset = UnifiedDataset(
101
+ base_path=args.dataset_base_path,
102
+ metadata_path=args.dataset_metadata_path,
103
+ repeat=args.dataset_repeat,
104
+ data_file_keys=args.data_file_keys.split(","),
105
+ main_data_operator=UnifiedDataset.default_image_operator(
106
+ base_path=args.dataset_base_path,
107
+ max_pixels=args.max_pixels,
108
+ height=args.height,
109
+ width=args.width,
110
+ height_division_factor=32,
111
+ width_division_factor=32,
112
+ )
113
+ )
114
+ model = StableDiffusionTrainingModule(
115
+ model_paths=args.model_paths,
116
+ model_id_with_origin_paths=args.model_id_with_origin_paths,
117
+ tokenizer_path=args.tokenizer_path,
118
+ trainable_models=args.trainable_models,
119
+ lora_base_model=args.lora_base_model,
120
+ lora_target_modules=args.lora_target_modules,
121
+ lora_rank=args.lora_rank,
122
+ lora_checkpoint=args.lora_checkpoint,
123
+ preset_lora_path=args.preset_lora_path,
124
+ preset_lora_model=args.preset_lora_model,
125
+ use_gradient_checkpointing=args.use_gradient_checkpointing,
126
+ use_gradient_checkpointing_offload=args.use_gradient_checkpointing_offload,
127
+ extra_inputs=args.extra_inputs,
128
+ fp8_models=args.fp8_models,
129
+ offload_models=args.offload_models,
130
+ quant_options=args.quant_options,
131
+ resume_from_checkpoint=args.resume_from_checkpoint,
132
+ remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
133
+ task=args.task,
134
+ device="cpu" if args.enable_model_cpu_offload else accelerator.device,
135
+ )
136
+ model_logger = ModelLogger(
137
+ args.output_path,
138
+ remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
139
+ enable_tensorboard_log=args.enable_tensorboard_log,
140
+ enable_swanlab_log=args.enable_swanlab_log,
141
+ swanlab_project=args.swanlab_project,
142
+ enable_wandb_log=args.enable_wandb_log,
143
+ wandb_project=args.wandb_project,
144
+ enable_csv_log=args.enable_csv_log,
145
+ )
146
+ launcher_map = {
147
+ "sft:data_process": launch_data_process_task,
148
+ "direct_distill:data_process": launch_data_process_task,
149
+ "sft": launch_training_task,
150
+ "sft:train": launch_training_task,
151
+ "direct_distill": launch_training_task,
152
+ "direct_distill:train": launch_training_task,
153
+ }
154
+ launcher_map[args.task](accelerator, dataset, model, model_logger, args=args)
examples/stable_diffusion/model_training/validate_full/stable-diffusion-v1-5.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline, ModelConfig
2
+ from diffsynth.core import load_state_dict
3
+ import torch
4
+
5
+ pipe = StableDiffusionPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
10
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
11
+ ],
12
+ tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
13
+ )
14
+ state_dict = load_state_dict("./models/train/stable-diffusion-v1-5_full/epoch-1.safetensors", torch_dtype=torch.float32)
15
+ pipe.unet.load_state_dict(state_dict)
16
+
17
+ image = pipe(
18
+ prompt="a dog",
19
+ negative_prompt="blurry, low quality, deformed",
20
+ cfg_scale=7.5,
21
+ height=512,
22
+ width=512,
23
+ seed=42,
24
+ rand_device="cuda",
25
+ num_inference_steps=50,
26
+ )
27
+ image.save("image_stable-diffusion-v1-5_full.jpg")
examples/stable_diffusion/model_training/validate_lora/stable-diffusion-v1-5.py ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
4
+
5
+ pipe = StableDiffusionPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
10
+ ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
11
+ ],
12
+ tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
13
+ )
14
+ pipe.load_lora(pipe.unet, "models/train/stable-diffusion-v1-5_lora/epoch-4.safetensors")
15
+
16
+ image = pipe(
17
+ prompt="a dog",
18
+ negative_prompt="blurry, low quality, deformed",
19
+ cfg_scale=7.5,
20
+ height=512,
21
+ width=512,
22
+ seed=42,
23
+ rand_device="cuda",
24
+ num_inference_steps=50,
25
+ )
26
+ image.save("image_stable-diffusion-v1-5.jpg")
examples/stable_diffusion_xl/model_inference/stable-diffusion-xl-base-1.0.py ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
4
+
5
+ pipe = StableDiffusionXLPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
10
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
11
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
14
+ tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
15
+ )
16
+
17
+ image = pipe(
18
+ prompt="a photo of an astronaut riding a horse on mars",
19
+ negative_prompt="",
20
+ cfg_scale=5.0,
21
+ height=1024,
22
+ width=1024,
23
+ seed=42,
24
+ num_inference_steps=50,
25
+ )
26
+ image.save("image.jpg")
examples/stable_diffusion_xl/model_inference_low_vram/stable-diffusion-xl-base-1.0.py ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
4
+
5
+ vram_config = {
6
+ "offload_dtype": torch.float32,
7
+ "offload_device": "cpu",
8
+ "onload_dtype": torch.float32,
9
+ "onload_device": "cpu",
10
+ "preparing_dtype": torch.float32,
11
+ "preparing_device": "cuda",
12
+ "computation_dtype": torch.float32,
13
+ "computation_device": "cuda",
14
+ }
15
+ pipe = StableDiffusionXLPipeline.from_pretrained(
16
+ torch_dtype=torch.float32,
17
+ model_configs=[
18
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors", **vram_config),
19
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors", **vram_config),
20
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors", **vram_config),
21
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config),
22
+ ],
23
+ tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
24
+ tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
25
+ vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
26
+ )
27
+
28
+ image = pipe(
29
+ prompt="a photo of an astronaut riding a horse on mars",
30
+ negative_prompt="",
31
+ cfg_scale=5.0,
32
+ height=1024,
33
+ width=1024,
34
+ seed=42,
35
+ num_inference_steps=50,
36
+ )
37
+ image.save("image.jpg")
examples/stable_diffusion_xl/model_training/full/stable-diffusion-xl-base-1.0.sh ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion_xl/stable-diffusion-xl-base-1.0/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ accelerate launch examples/stable_diffusion_xl/model_training/train.py \
4
+ --dataset_base_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0 \
5
+ --dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0/metadata.csv \
6
+ --height 1024 \
7
+ --width 1024 \
8
+ --dataset_repeat 10 \
9
+ --model_id_with_origin_paths "stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors" \
10
+ --learning_rate 1e-5 \
11
+ --num_epochs 2 \
12
+ --trainable_models "unet" \
13
+ --remove_prefix_in_ckpt "pipe.unet." \
14
+ --output_path "./models/train/stable-diffusion-xl-base-1.0_full" \
15
+ --use_gradient_checkpointing
examples/stable_diffusion_xl/model_training/lora/stable-diffusion-xl-base-1.0.sh ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion_xl/stable-diffusion-xl-base-1.0/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ accelerate launch examples/stable_diffusion_xl/model_training/train.py \
4
+ --dataset_base_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0 \
5
+ --dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0/metadata.csv \
6
+ --height 1024 \
7
+ --width 1024 \
8
+ --dataset_repeat 10 \
9
+ --model_id_with_origin_paths "stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors" \
10
+ --learning_rate 1e-4 \
11
+ --num_epochs 5 \
12
+ --remove_prefix_in_ckpt "pipe.unet." \
13
+ --output_path "./models/train/stable-diffusion-xl-base-1.0_lora" \
14
+ --lora_base_model "unet" \
15
+ --lora_target_modules "mid_block.attentions.0.proj_in,mid_block.attentions.0.proj_out,down_blocks.1.attentions.0.proj_in,down_blocks.1.attentions.0.proj_out,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.1.attentions.1.proj_in,down_blocks.1.attentions.1.proj_out,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.proj_in,down_blocks.2.attentions.0.proj_out,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.2,down_blocks.2.attentions.1.proj_in,down_blocks.2.attentions.1.proj_out,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.2,mid_block.attentions.0.transformer_blocks.0.attn1.to_k,mid_block.attentions.0.transformer_blocks.0.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn1.to_q,mid_block.attentions.0.transformer_blocks.0.attn1.to_v,mid_block.attentions.0.transformer_blocks.0.attn2.to_k,mid_block.attentions.0.transformer_blocks.0.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn2.to_q,mid_block.attentions.0.transformer_blocks.0.attn2.to_v,mid_block.attentions.0.transformer_blocks.0.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.0.ff.net.2,mid_block.attentions.0.transformer_blocks.1.attn1.to_k,mid_block.attentions.0.transformer_blocks.1.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn1.to_q,mid_block.attentions.0.transformer_blocks.1.attn1.to_v,mid_block.attentions.0.transformer_blocks.1.attn2.to_k,mid_block.attentions.0.transformer_blocks.1.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn2.to_q,mid_block.attentions.0.transformer_blocks.1.attn2.to_v,mid_block.attentions.0.transformer_blocks.1.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.1.ff.net.2,mid_block.attentions.0.transformer_blocks.2.attn1.to_k,mid_block.attentions.0.transformer_blocks.2.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn1.to_q,mid_block.attentions.0.transformer_blocks.2.attn1.to_v,mid_block.attentions.0.transformer_blocks.2.attn2.to_k,mid_block.attentions.0.transformer_blocks.2.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn2.to_q,mid_block.attentions.0.transformer_blocks.2.attn2.to_v,mid_block.attentions.0.transformer_blocks.2.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.2.ff.net.2,mid_block.attentions.0.transformer_blocks.3.attn1.to_k,mid_block.attentions.0.transformer_blocks.3.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn1.to_q,mid_block.attentions.0.transformer_blocks.3.attn1.to_v,mid_block.attentions.0.transformer_blocks.3.attn2.to_k,mid_block.attentions.0.transformer_blocks.3.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn2.to_q,mid_block.attentions.0.transformer_blocks.3.attn2.to_v,mid_block.attentions.0.transformer_blocks.3.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.3.ff.net.2,mid_block.attentions.0.transformer_blocks.4.attn1.to_k,mid_block.attentions.0.transformer_blocks.4.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn1.to_q,mid_block.attentions.0.transformer_blocks.4.attn1.to_v,mid_block.attentions.0.transformer_blocks.4.attn2.to_k,mid_block.attentions.0.transformer_blocks.4.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn2.to_q,mid_block.attentions.0.transformer_blocks.4.attn2.to_v,mid_block.attentions.0.transformer_blocks.4.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.4.ff.net.2,mid_block.attentions.0.transformer_blocks.5.attn1.to_k,mid_block.attentions.0.transformer_blocks.5.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn1.to_q,mid_block.attentions.0.transformer_blocks.5.attn1.to_v,mid_block.attentions.0.transformer_blocks.5.attn2.to_k,mid_block.attentions.0.transformer_blocks.5.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn2.to_q,mid_block.attentions.0.transformer_blocks.5.attn2.to_v,mid_block.attentions.0.transformer_blocks.5.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.5.ff.net.2,mid_block.attentions.0.transformer_blocks.6.attn1.to_k,mid_block.attentions.0.transformer_blocks.6.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn1.to_q,mid_block.attentions.0.transformer_blocks.6.attn1.to_v,mid_block.attentions.0.transformer_blocks.6.attn2.to_k,mid_block.attentions.0.transformer_blocks.6.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn2.to_q,mid_block.attentions.0.transformer_blocks.6.attn2.to_v,mid_block.attentions.0.transformer_blocks.6.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.6.ff.net.2,mid_block.attentions.0.transformer_blocks.7.attn1.to_k,mid_block.attentions.0.transformer_blocks.7.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn1.to_q,mid_block.attentions.0.transformer_blocks.7.attn1.to_v,mid_block.attentions.0.transformer_blocks.7.attn2.to_k,mid_block.attentions.0.transformer_blocks.7.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn2.to_q,mid_block.attentions.0.transformer_blocks.7.attn2.to_v,mid_block.attentions.0.transformer_blocks.7.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.7.ff.net.2,mid_block.attentions.0.transformer_blocks.8.attn1.to_k,mid_block.attentions.0.transformer_blocks.8.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn1.to_q,mid_block.attentions.0.transformer_blocks.8.attn1.to_v,mid_block.attentions.0.transformer_blocks.8.attn2.to_k,mid_block.attentions.0.transformer_blocks.8.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn2.to_q,mid_block.attentions.0.transformer_blocks.8.attn2.to_v,mid_block.attentions.0.transformer_blocks.8.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.8.ff.net.2,mid_block.attentions.0.transformer_blocks.9.attn1.to_k,mid_block.attentions.0.transformer_blocks.9.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn1.to_q,mid_block.attentions.0.transformer_blocks.9.attn1.to_v,mid_block.attentions.0.transformer_blocks.9.attn2.to_k,mid_block.attentions.0.transformer_blocks.9.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn2.to_q,mid_block.attentions.0.transformer_blocks.9.attn2.to_v,mid_block.attentions.0.transformer_blocks.9.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.0.proj_in,up_blocks.0.attentions.0.proj_out,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.1.proj_in,up_blocks.0.attentions.1.proj_out,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.2.proj_in,up_blocks.0.attentions.2.proj_out,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.2,up_blocks.1.attentions.0.proj_in,up_blocks.1.attentions.0.proj_out,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.1.proj_in,up_blocks.1.attentions.1.proj_out,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.2.proj_in,up_blocks.1.attentions.2.proj_out,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.2" \
16
+ --lora_rank 32 \
17
+ --use_gradient_checkpointing \
18
+ --align_to_opensource_format
examples/stable_diffusion_xl/model_training/special/split_training/stable-diffusion-xl-base-1.0.sh ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion_xl/stable-diffusion-xl-base-1.0/*" --local_dir ./data/diffsynth_example_dataset
2
+
3
+ # Stage 1: cache deterministic preprocessing outputs.
4
+ accelerate launch examples/stable_diffusion_xl/model_training/train.py \
5
+ --dataset_base_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0 \
6
+ --dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0/metadata.csv \
7
+ --height 1024 \
8
+ --width 1024 \
9
+ --dataset_repeat 1 \
10
+ --model_id_with_origin_paths stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors \
11
+ --learning_rate 1e-4 \
12
+ --num_epochs 5 \
13
+ --remove_prefix_in_ckpt pipe.unet. \
14
+ --output_path ./models/train/stable-diffusion-xl-base-1.0_split_cache \
15
+ --lora_base_model unet \
16
+ --lora_target_modules mid_block.attentions.0.proj_in,mid_block.attentions.0.proj_out,down_blocks.1.attentions.0.proj_in,down_blocks.1.attentions.0.proj_out,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.1.attentions.1.proj_in,down_blocks.1.attentions.1.proj_out,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.proj_in,down_blocks.2.attentions.0.proj_out,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.2,down_blocks.2.attentions.1.proj_in,down_blocks.2.attentions.1.proj_out,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.2,mid_block.attentions.0.transformer_blocks.0.attn1.to_k,mid_block.attentions.0.transformer_blocks.0.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn1.to_q,mid_block.attentions.0.transformer_blocks.0.attn1.to_v,mid_block.attentions.0.transformer_blocks.0.attn2.to_k,mid_block.attentions.0.transformer_blocks.0.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn2.to_q,mid_block.attentions.0.transformer_blocks.0.attn2.to_v,mid_block.attentions.0.transformer_blocks.0.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.0.ff.net.2,mid_block.attentions.0.transformer_blocks.1.attn1.to_k,mid_block.attentions.0.transformer_blocks.1.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn1.to_q,mid_block.attentions.0.transformer_blocks.1.attn1.to_v,mid_block.attentions.0.transformer_blocks.1.attn2.to_k,mid_block.attentions.0.transformer_blocks.1.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn2.to_q,mid_block.attentions.0.transformer_blocks.1.attn2.to_v,mid_block.attentions.0.transformer_blocks.1.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.1.ff.net.2,mid_block.attentions.0.transformer_blocks.2.attn1.to_k,mid_block.attentions.0.transformer_blocks.2.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn1.to_q,mid_block.attentions.0.transformer_blocks.2.attn1.to_v,mid_block.attentions.0.transformer_blocks.2.attn2.to_k,mid_block.attentions.0.transformer_blocks.2.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn2.to_q,mid_block.attentions.0.transformer_blocks.2.attn2.to_v,mid_block.attentions.0.transformer_blocks.2.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.2.ff.net.2,mid_block.attentions.0.transformer_blocks.3.attn1.to_k,mid_block.attentions.0.transformer_blocks.3.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn1.to_q,mid_block.attentions.0.transformer_blocks.3.attn1.to_v,mid_block.attentions.0.transformer_blocks.3.attn2.to_k,mid_block.attentions.0.transformer_blocks.3.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn2.to_q,mid_block.attentions.0.transformer_blocks.3.attn2.to_v,mid_block.attentions.0.transformer_blocks.3.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.3.ff.net.2,mid_block.attentions.0.transformer_blocks.4.attn1.to_k,mid_block.attentions.0.transformer_blocks.4.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn1.to_q,mid_block.attentions.0.transformer_blocks.4.attn1.to_v,mid_block.attentions.0.transformer_blocks.4.attn2.to_k,mid_block.attentions.0.transformer_blocks.4.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn2.to_q,mid_block.attentions.0.transformer_blocks.4.attn2.to_v,mid_block.attentions.0.transformer_blocks.4.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.4.ff.net.2,mid_block.attentions.0.transformer_blocks.5.attn1.to_k,mid_block.attentions.0.transformer_blocks.5.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn1.to_q,mid_block.attentions.0.transformer_blocks.5.attn1.to_v,mid_block.attentions.0.transformer_blocks.5.attn2.to_k,mid_block.attentions.0.transformer_blocks.5.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn2.to_q,mid_block.attentions.0.transformer_blocks.5.attn2.to_v,mid_block.attentions.0.transformer_blocks.5.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.5.ff.net.2,mid_block.attentions.0.transformer_blocks.6.attn1.to_k,mid_block.attentions.0.transformer_blocks.6.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn1.to_q,mid_block.attentions.0.transformer_blocks.6.attn1.to_v,mid_block.attentions.0.transformer_blocks.6.attn2.to_k,mid_block.attentions.0.transformer_blocks.6.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn2.to_q,mid_block.attentions.0.transformer_blocks.6.attn2.to_v,mid_block.attentions.0.transformer_blocks.6.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.6.ff.net.2,mid_block.attentions.0.transformer_blocks.7.attn1.to_k,mid_block.attentions.0.transformer_blocks.7.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn1.to_q,mid_block.attentions.0.transformer_blocks.7.attn1.to_v,mid_block.attentions.0.transformer_blocks.7.attn2.to_k,mid_block.attentions.0.transformer_blocks.7.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn2.to_q,mid_block.attentions.0.transformer_blocks.7.attn2.to_v,mid_block.attentions.0.transformer_blocks.7.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.7.ff.net.2,mid_block.attentions.0.transformer_blocks.8.attn1.to_k,mid_block.attentions.0.transformer_blocks.8.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn1.to_q,mid_block.attentions.0.transformer_blocks.8.attn1.to_v,mid_block.attentions.0.transformer_blocks.8.attn2.to_k,mid_block.attentions.0.transformer_blocks.8.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn2.to_q,mid_block.attentions.0.transformer_blocks.8.attn2.to_v,mid_block.attentions.0.transformer_blocks.8.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.8.ff.net.2,mid_block.attentions.0.transformer_blocks.9.attn1.to_k,mid_block.attentions.0.transformer_blocks.9.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn1.to_q,mid_block.attentions.0.transformer_blocks.9.attn1.to_v,mid_block.attentions.0.transformer_blocks.9.attn2.to_k,mid_block.attentions.0.transformer_blocks.9.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn2.to_q,mid_block.attentions.0.transformer_blocks.9.attn2.to_v,mid_block.attentions.0.transformer_blocks.9.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.0.proj_in,up_blocks.0.attentions.0.proj_out,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.1.proj_in,up_blocks.0.attentions.1.proj_out,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.2.proj_in,up_blocks.0.attentions.2.proj_out,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.2,up_blocks.1.attentions.0.proj_in,up_blocks.1.attentions.0.proj_out,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.1.proj_in,up_blocks.1.attentions.1.proj_out,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.2.proj_in,up_blocks.1.attentions.2.proj_out,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.2 \
17
+ --lora_rank 32 \
18
+ --use_gradient_checkpointing \
19
+ --align_to_opensource_format \
20
+ --offload_models stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors \
21
+ --task sft:data_process
22
+
23
+ # Stage 2: train LoRA from the cached dataset.
24
+ accelerate launch examples/stable_diffusion_xl/model_training/train.py \
25
+ --dataset_base_path ./models/train/stable-diffusion-xl-base-1.0_split_cache \
26
+ --height 1024 \
27
+ --width 1024 \
28
+ --dataset_repeat 10 \
29
+ --model_id_with_origin_paths stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors \
30
+ --learning_rate 1e-4 \
31
+ --num_epochs 5 \
32
+ --remove_prefix_in_ckpt pipe.unet. \
33
+ --output_path ./models/train/stable-diffusion-xl-base-1.0_split \
34
+ --lora_base_model unet \
35
+ --lora_target_modules mid_block.attentions.0.proj_in,mid_block.attentions.0.proj_out,down_blocks.1.attentions.0.proj_in,down_blocks.1.attentions.0.proj_out,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.1.attentions.1.proj_in,down_blocks.1.attentions.1.proj_out,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.proj_in,down_blocks.2.attentions.0.proj_out,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.2,down_blocks.2.attentions.1.proj_in,down_blocks.2.attentions.1.proj_out,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.2,mid_block.attentions.0.transformer_blocks.0.attn1.to_k,mid_block.attentions.0.transformer_blocks.0.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn1.to_q,mid_block.attentions.0.transformer_blocks.0.attn1.to_v,mid_block.attentions.0.transformer_blocks.0.attn2.to_k,mid_block.attentions.0.transformer_blocks.0.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn2.to_q,mid_block.attentions.0.transformer_blocks.0.attn2.to_v,mid_block.attentions.0.transformer_blocks.0.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.0.ff.net.2,mid_block.attentions.0.transformer_blocks.1.attn1.to_k,mid_block.attentions.0.transformer_blocks.1.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn1.to_q,mid_block.attentions.0.transformer_blocks.1.attn1.to_v,mid_block.attentions.0.transformer_blocks.1.attn2.to_k,mid_block.attentions.0.transformer_blocks.1.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn2.to_q,mid_block.attentions.0.transformer_blocks.1.attn2.to_v,mid_block.attentions.0.transformer_blocks.1.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.1.ff.net.2,mid_block.attentions.0.transformer_blocks.2.attn1.to_k,mid_block.attentions.0.transformer_blocks.2.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn1.to_q,mid_block.attentions.0.transformer_blocks.2.attn1.to_v,mid_block.attentions.0.transformer_blocks.2.attn2.to_k,mid_block.attentions.0.transformer_blocks.2.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn2.to_q,mid_block.attentions.0.transformer_blocks.2.attn2.to_v,mid_block.attentions.0.transformer_blocks.2.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.2.ff.net.2,mid_block.attentions.0.transformer_blocks.3.attn1.to_k,mid_block.attentions.0.transformer_blocks.3.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn1.to_q,mid_block.attentions.0.transformer_blocks.3.attn1.to_v,mid_block.attentions.0.transformer_blocks.3.attn2.to_k,mid_block.attentions.0.transformer_blocks.3.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn2.to_q,mid_block.attentions.0.transformer_blocks.3.attn2.to_v,mid_block.attentions.0.transformer_blocks.3.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.3.ff.net.2,mid_block.attentions.0.transformer_blocks.4.attn1.to_k,mid_block.attentions.0.transformer_blocks.4.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn1.to_q,mid_block.attentions.0.transformer_blocks.4.attn1.to_v,mid_block.attentions.0.transformer_blocks.4.attn2.to_k,mid_block.attentions.0.transformer_blocks.4.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn2.to_q,mid_block.attentions.0.transformer_blocks.4.attn2.to_v,mid_block.attentions.0.transformer_blocks.4.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.4.ff.net.2,mid_block.attentions.0.transformer_blocks.5.attn1.to_k,mid_block.attentions.0.transformer_blocks.5.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn1.to_q,mid_block.attentions.0.transformer_blocks.5.attn1.to_v,mid_block.attentions.0.transformer_blocks.5.attn2.to_k,mid_block.attentions.0.transformer_blocks.5.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn2.to_q,mid_block.attentions.0.transformer_blocks.5.attn2.to_v,mid_block.attentions.0.transformer_blocks.5.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.5.ff.net.2,mid_block.attentions.0.transformer_blocks.6.attn1.to_k,mid_block.attentions.0.transformer_blocks.6.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn1.to_q,mid_block.attentions.0.transformer_blocks.6.attn1.to_v,mid_block.attentions.0.transformer_blocks.6.attn2.to_k,mid_block.attentions.0.transformer_blocks.6.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn2.to_q,mid_block.attentions.0.transformer_blocks.6.attn2.to_v,mid_block.attentions.0.transformer_blocks.6.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.6.ff.net.2,mid_block.attentions.0.transformer_blocks.7.attn1.to_k,mid_block.attentions.0.transformer_blocks.7.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn1.to_q,mid_block.attentions.0.transformer_blocks.7.attn1.to_v,mid_block.attentions.0.transformer_blocks.7.attn2.to_k,mid_block.attentions.0.transformer_blocks.7.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn2.to_q,mid_block.attentions.0.transformer_blocks.7.attn2.to_v,mid_block.attentions.0.transformer_blocks.7.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.7.ff.net.2,mid_block.attentions.0.transformer_blocks.8.attn1.to_k,mid_block.attentions.0.transformer_blocks.8.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn1.to_q,mid_block.attentions.0.transformer_blocks.8.attn1.to_v,mid_block.attentions.0.transformer_blocks.8.attn2.to_k,mid_block.attentions.0.transformer_blocks.8.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn2.to_q,mid_block.attentions.0.transformer_blocks.8.attn2.to_v,mid_block.attentions.0.transformer_blocks.8.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.8.ff.net.2,mid_block.attentions.0.transformer_blocks.9.attn1.to_k,mid_block.attentions.0.transformer_blocks.9.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn1.to_q,mid_block.attentions.0.transformer_blocks.9.attn1.to_v,mid_block.attentions.0.transformer_blocks.9.attn2.to_k,mid_block.attentions.0.transformer_blocks.9.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn2.to_q,mid_block.attentions.0.transformer_blocks.9.attn2.to_v,mid_block.attentions.0.transformer_blocks.9.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.0.proj_in,up_blocks.0.attentions.0.proj_out,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.1.proj_in,up_blocks.0.attentions.1.proj_out,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.2.proj_in,up_blocks.0.attentions.2.proj_out,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.2,up_blocks.1.attentions.0.proj_in,up_blocks.1.attentions.0.proj_out,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.1.proj_in,up_blocks.1.attentions.1.proj_out,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.2.proj_in,up_blocks.1.attentions.2.proj_out,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.2 \
36
+ --lora_rank 32 \
37
+ --use_gradient_checkpointing \
38
+ --align_to_opensource_format \
39
+ --offload_models stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors \
40
+ --task sft:train
examples/stable_diffusion_xl/model_training/special/split_training/validate.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
4
+
5
+ pipe = StableDiffusionXLPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
10
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
11
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
14
+ tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
15
+ )
16
+ pipe.load_lora(pipe.unet, './models/train/stable-diffusion-xl-base-1.0_split/epoch-4.safetensors')
17
+
18
+ image = pipe(
19
+ prompt="a dog",
20
+ negative_prompt="",
21
+ cfg_scale=7.0,
22
+ height=1024,
23
+ width=1024,
24
+ seed=42,
25
+ num_inference_steps=50,
26
+ )
27
+ image.save('split_training_stable-diffusion-xl.jpg')
examples/stable_diffusion_xl/model_training/train.py ADDED
@@ -0,0 +1,159 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch, os, argparse, accelerate
2
+ from diffsynth.core import UnifiedDataset
3
+ from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline, ModelConfig
4
+ from diffsynth.diffusion import *
5
+ from diffsynth.utils.lora.sdxl import SdxlLoRAConverter
6
+ os.environ["TOKENIZERS_PARALLELISM"] = "false"
7
+
8
+
9
+ class StableDiffusionXLTrainingModule(DiffusionTrainingModule):
10
+ def __init__(
11
+ self,
12
+ model_paths=None, model_id_with_origin_paths=None,
13
+ tokenizer_path=None,
14
+ trainable_models=None,
15
+ lora_base_model=None, lora_target_modules="", lora_rank=32, lora_checkpoint=None,
16
+ preset_lora_path=None, preset_lora_model=None,
17
+ use_gradient_checkpointing=True,
18
+ use_gradient_checkpointing_offload=False,
19
+ extra_inputs=None,
20
+ fp8_models=None,
21
+ offload_models=None,
22
+ quant_options=None,
23
+ resume_from_checkpoint=None, remove_prefix_in_ckpt=None,
24
+ device="cpu",
25
+ task="sft",
26
+ ):
27
+ super().__init__()
28
+ # Load models
29
+ model_configs = self.parse_model_configs(model_paths, model_id_with_origin_paths, fp8_models=fp8_models, offload_models=offload_models, quant_options=quant_options, device=device)
30
+ tokenizer_config = self.parse_path_or_model_id(tokenizer_path, ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"))
31
+ tokenizer_2_config = self.parse_path_or_model_id(tokenizer_path, ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"))
32
+ self.pipe = StableDiffusionXLPipeline.from_pretrained(torch_dtype=torch.float32, device=device, model_configs=model_configs, tokenizer_config=tokenizer_config, tokenizer_2_config=tokenizer_2_config)
33
+ self.pipe = self.split_pipeline_units(task, self.pipe, trainable_models, lora_base_model)
34
+ self.resume_from_checkpoint(resume_from_checkpoint, remove_prefix_in_ckpt)
35
+
36
+ # Training mode
37
+ self.switch_pipe_to_training_mode(
38
+ self.pipe, trainable_models,
39
+ lora_base_model, lora_target_modules, lora_rank, lora_checkpoint,
40
+ preset_lora_path, preset_lora_model,
41
+ task=task,
42
+ )
43
+
44
+ # Other configs
45
+ self.use_gradient_checkpointing = use_gradient_checkpointing
46
+ self.use_gradient_checkpointing_offload = use_gradient_checkpointing_offload
47
+ self.extra_inputs = extra_inputs.split(",") if extra_inputs is not None else []
48
+ self.fp8_models = fp8_models
49
+ self.task = task
50
+ self.task_to_loss = {
51
+ "sft:data_process": lambda pipe, *args: args,
52
+ "direct_distill:data_process": lambda pipe, *args: args,
53
+ "sft": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
54
+ "sft:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
55
+ "direct_distill": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
56
+ "direct_distill:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
57
+ }
58
+
59
+ def get_pipeline_inputs(self, data):
60
+ inputs_posi = {"prompt": data["prompt"]}
61
+ inputs_nega = {"negative_prompt": ""}
62
+ inputs_shared = {
63
+ # Assume you are using this pipeline for inference,
64
+ # please fill in the input parameters.
65
+ "input_image": data["image"],
66
+ "height": data["image"].size[1],
67
+ "width": data["image"].size[0],
68
+ # Please do not modify the following parameters
69
+ # unless you clearly know what this will cause.
70
+ "cfg_scale": 1,
71
+ "rand_device": self.pipe.device,
72
+ "use_gradient_checkpointing": self.use_gradient_checkpointing,
73
+ "use_gradient_checkpointing_offload": self.use_gradient_checkpointing_offload,
74
+ }
75
+ inputs_shared = self.parse_extra_inputs(data, self.extra_inputs, inputs_shared)
76
+ return inputs_shared, inputs_posi, inputs_nega
77
+
78
+ def forward(self, data, inputs=None):
79
+ if inputs is None: inputs = self.get_pipeline_inputs(data)
80
+ inputs = self.transfer_data_to_device(inputs, self.pipe.device, self.pipe.torch_dtype)
81
+ for unit in self.pipe.units:
82
+ inputs = self.pipe.unit_runner(unit, self.pipe, *inputs)
83
+ loss = self.task_to_loss[self.task](self.pipe, *inputs)
84
+ return loss
85
+
86
+
87
+ def parser():
88
+ parser = argparse.ArgumentParser(description="Simple example of a training script.")
89
+ parser = add_general_config(parser)
90
+ parser = add_image_size_config(parser)
91
+ parser.add_argument("--tokenizer_path", type=str, default=None, help="Path to tokenizer.")
92
+ parser.add_argument("--tokenizer_2_path", type=str, default=None, help="Path to tokenizer 2.")
93
+ parser.add_argument("--align_to_opensource_format", default=False, action="store_true", help="Whether to align the lora format to opensource format.")
94
+ return parser
95
+
96
+
97
+ if __name__ == "__main__":
98
+ parser = parser()
99
+ args = parser.parse_args()
100
+ accelerator = accelerate.Accelerator(
101
+ gradient_accumulation_steps=args.gradient_accumulation_steps,
102
+ kwargs_handlers=[accelerate.DistributedDataParallelKwargs(find_unused_parameters=args.find_unused_parameters)],
103
+ )
104
+ dataset = UnifiedDataset(
105
+ base_path=args.dataset_base_path,
106
+ metadata_path=args.dataset_metadata_path,
107
+ repeat=args.dataset_repeat,
108
+ data_file_keys=args.data_file_keys.split(","),
109
+ main_data_operator=UnifiedDataset.default_image_operator(
110
+ base_path=args.dataset_base_path,
111
+ max_pixels=args.max_pixels,
112
+ height=args.height,
113
+ width=args.width,
114
+ height_division_factor=32,
115
+ width_division_factor=32,
116
+ )
117
+ )
118
+ model = StableDiffusionXLTrainingModule(
119
+ model_paths=args.model_paths,
120
+ model_id_with_origin_paths=args.model_id_with_origin_paths,
121
+ tokenizer_path=args.tokenizer_path,
122
+ trainable_models=args.trainable_models,
123
+ lora_base_model=args.lora_base_model,
124
+ lora_target_modules=args.lora_target_modules,
125
+ lora_rank=args.lora_rank,
126
+ lora_checkpoint=args.lora_checkpoint,
127
+ preset_lora_path=args.preset_lora_path,
128
+ preset_lora_model=args.preset_lora_model,
129
+ use_gradient_checkpointing=args.use_gradient_checkpointing,
130
+ use_gradient_checkpointing_offload=args.use_gradient_checkpointing_offload,
131
+ extra_inputs=args.extra_inputs,
132
+ fp8_models=args.fp8_models,
133
+ offload_models=args.offload_models,
134
+ quant_options=args.quant_options,
135
+ resume_from_checkpoint=args.resume_from_checkpoint,
136
+ remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
137
+ task=args.task,
138
+ device="cpu" if args.enable_model_cpu_offload else accelerator.device,
139
+ )
140
+ model_logger = ModelLogger(
141
+ args.output_path,
142
+ remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
143
+ state_dict_converter=SdxlLoRAConverter.align_to_opensource_format if args.align_to_opensource_format else lambda x:x,
144
+ enable_tensorboard_log=args.enable_tensorboard_log,
145
+ enable_swanlab_log=args.enable_swanlab_log,
146
+ swanlab_project=args.swanlab_project,
147
+ enable_wandb_log=args.enable_wandb_log,
148
+ wandb_project=args.wandb_project,
149
+ enable_csv_log=args.enable_csv_log,
150
+ )
151
+ launcher_map = {
152
+ "sft:data_process": launch_data_process_task,
153
+ "direct_distill:data_process": launch_data_process_task,
154
+ "sft": launch_training_task,
155
+ "sft:train": launch_training_task,
156
+ "direct_distill": launch_training_task,
157
+ "direct_distill:train": launch_training_task,
158
+ }
159
+ launcher_map[args.task](accelerator, dataset, model, model_logger, args=args)
examples/stable_diffusion_xl/model_training/validate_full/stable-diffusion-xl-base-1.0.py ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline, ModelConfig
2
+ from diffsynth.core import load_state_dict
3
+ import torch
4
+
5
+ pipe = StableDiffusionXLPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
10
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
11
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
14
+ tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
15
+ )
16
+ state_dict = load_state_dict("./models/train/stable-diffusion-xl-base-1.0_full/epoch-1.safetensors", torch_dtype=torch.float32)
17
+ pipe.unet.load_state_dict(state_dict)
18
+
19
+ image = pipe(
20
+ prompt="a dog",
21
+ negative_prompt="",
22
+ cfg_scale=7.0,
23
+ height=1024,
24
+ width=1024,
25
+ seed=42,
26
+ num_inference_steps=50,
27
+ )
28
+ image.save("image_stable-diffusion-xl-base-1.0_full.jpg")
examples/stable_diffusion_xl/model_training/validate_lora/stable-diffusion-xl-base-1.0.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.core import ModelConfig
3
+ from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
4
+
5
+ pipe = StableDiffusionXLPipeline.from_pretrained(
6
+ torch_dtype=torch.float32,
7
+ model_configs=[
8
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
9
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
10
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
11
+ ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
12
+ ],
13
+ tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
14
+ tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
15
+ )
16
+ pipe.load_lora(pipe.unet, "models/train/stable-diffusion-xl-base-1.0_lora/epoch-4.safetensors")
17
+
18
+ image = pipe(
19
+ prompt="a dog",
20
+ negative_prompt="",
21
+ cfg_scale=7.0,
22
+ height=1024,
23
+ width=1024,
24
+ seed=42,
25
+ num_inference_steps=50,
26
+ )
27
+ image.save("image_stable-diffusion-xl-base-1.0.jpg")
examples/wanvideo/README.md ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ English Document: https://diffsynth-studio-doc.readthedocs.io/en/latest/Model_Details/Wan.html
2
+
3
+ 中文文档:https://diffsynth-studio-doc.readthedocs.io/zh-cn/latest/Model_Details/Wan.html
examples/wanvideo/acceleration/Wan2.2-Animate-2-14B-usp.py ADDED
@@ -0,0 +1,117 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ import torch.distributed as dist
3
+ from PIL import Image
4
+ from diffsynth.utils.data import save_video, VideoData
5
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
6
+ from modelscope import dataset_snapshot_download
7
+
8
+ vram_config = {
9
+ "offload_dtype": torch.bfloat16,
10
+ "offload_device": "cpu",
11
+ "onload_dtype": torch.bfloat16,
12
+ "onload_device": "cuda",
13
+ "preparing_dtype": torch.bfloat16,
14
+ "preparing_device": "cuda",
15
+ "computation_dtype": torch.bfloat16,
16
+ "computation_device": "cuda",
17
+ }
18
+
19
+ pipe = WanVideoPipeline.from_pretrained(
20
+ torch_dtype=torch.bfloat16,
21
+ device="cuda",
22
+ use_usp=True,
23
+ model_configs=[
24
+ ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="wan_animate_2/wan_animate_2_bf16.safetensors", **vram_config),
25
+ ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="videomodel/Wan-AI/models_t5_umt5-xxl-enc-bf16.pth", **vram_config),
26
+ ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="videomodel/Wan-AI/models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth", **vram_config),
27
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="Wan2.1_VAE.pth", **vram_config),
28
+ ],
29
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="videomodel/Wan-AI/umt5-xxl/"),
30
+ )
31
+
32
+ # Character animation: reference image (identity) + reference video (motion) -> animated video.
33
+ dataset_snapshot_download(
34
+ "DiffSynth-Studio/diffsynth_example_dataset",
35
+ local_dir="data/diffsynth_example_dataset",
36
+ allow_file_pattern="wanvideo/Wan2.2-Animate-2-14B/*"
37
+ )
38
+ reference_image = Image.open("data/diffsynth_example_dataset/wanvideo/Wan2.2-Animate-2-14B/refimage.jpg").convert("RGB")
39
+ reference_video = VideoData("data/diffsynth_example_dataset/wanvideo/Wan2.2-Animate-2-14B/refvideo.mp4").raw_data()
40
+ # Example 1: single-clip generation
41
+ num_frames = 81
42
+ video = pipe(
43
+ prompt="人物外观描述:一名长黑发女性,穿着白色半透明蕾丝长袖上衣,衣身带有花卉刺绣,下身搭配白色百褶短裙和黑色腰带,脚穿米白色厚底运动鞋。 背景描述:背景为现代室内空间,墙面和柜体以浅灰色为主,后方设有两扇深色落地窗或玻璃门,顶部安装长条形灯具,中央有一块浅色长方形台面。",
44
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
45
+ animate2_prompt_ref="视频中的人在做动作,背景静止",
46
+ animate2_reference_image=reference_image,
47
+ animate2_reference_video=reference_video[:num_frames],
48
+ animate2_offload_kv=True,
49
+ num_frames=num_frames, height=1280, width=720,
50
+ num_inference_steps=40, cfg_scale=3.0,
51
+ seed=0, tiled=True,
52
+ )
53
+ if dist.get_rank() == 0:
54
+ save_video(video, "video_Wan2.2-Animate-2-14B.mp4", fps=24, quality=5)
55
+
56
+
57
+ # Example 2: multi-clip long-video generation
58
+ def generate_long_video(pipe, reference_image, cond_images, clip_len, first_num=1, **kwargs):
59
+ assert clip_len > first_num, "clip_len must be greater than first_num"
60
+
61
+ def zigzag_padding(array, target_len):
62
+ if len(array) == 1:
63
+ return [array[0]] * target_len
64
+ idx, flip, out = 0, False, []
65
+ while len(out) < target_len:
66
+ out.append(array[idx])
67
+ idx += -1 if flip else 1
68
+ if idx == 0 or idx == len(array) - 1:
69
+ flip = not flip
70
+ return out[:target_len]
71
+
72
+ real_len = len(cond_images)
73
+ if real_len == 0:
74
+ return []
75
+ step = clip_len - first_num
76
+ # Precompute clip count so clips of `clip_len` stepping by `step` tile the (padded) driving video.
77
+ num_clips = 1 if real_len <= clip_len else (real_len - clip_len + step - 1) // step + 1
78
+ target_len = clip_len + (num_clips - 1) * step
79
+ if real_len < target_len:
80
+ cond_images = zigzag_padding(cond_images, target_len)
81
+
82
+ all_frames = []
83
+ prev_tail = None
84
+ for i in range(num_clips):
85
+ start = i * step
86
+ seg_driving = cond_images[start:start + clip_len]
87
+ seg_out = pipe(
88
+ animate2_reference_image=reference_image,
89
+ animate2_reference_video=seg_driving,
90
+ animate2_refert_images=None if i == 0 else prev_tail,
91
+ num_frames=clip_len,
92
+ **kwargs,
93
+ )
94
+ prev_tail = seg_out[-first_num:]
95
+ if i != 0:
96
+ seg_out = seg_out[first_num:]
97
+ all_frames.extend(seg_out)
98
+ return all_frames[:real_len]
99
+
100
+
101
+ clip_len = 81
102
+ long_video = generate_long_video(
103
+ pipe,
104
+ reference_image=reference_image,
105
+ cond_images=reference_video,
106
+ clip_len=clip_len,
107
+ first_num=1,
108
+ prompt="人物外观描述:一名长黑发女性,穿着白色半透明蕾丝长袖上衣,衣身带有花卉刺绣,下身搭配白色百褶短裙和黑色腰带,脚穿米白色厚底运动鞋。 背景描述:背景为现代室内空间,墙面和柜体以浅灰色为主,后方设有两扇深色落地窗或玻璃门,顶部安装长条形灯具,中央有一块浅色长方形台面。",
109
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
110
+ animate2_prompt_ref="视频中的人在做动作,背景静止",
111
+ animate2_offload_kv=True,
112
+ height=1280, width=720,
113
+ num_inference_steps=40, cfg_scale=3.0,
114
+ seed=0, tiled=True,
115
+ )
116
+ if dist.get_rank() == 0:
117
+ save_video(long_video, "video_Wan2.2-Animate-2-14B-long.mp4", fps=24, quality=5)
examples/wanvideo/acceleration/unified_sequence_parallel.py ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ import torch.distributed as dist
6
+
7
+ pipe = WanVideoPipeline.from_pretrained(
8
+ torch_dtype=torch.bfloat16,
9
+ device="cuda",
10
+ use_usp=True,
11
+ model_configs=[
12
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ],
16
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
17
+ )
18
+
19
+ # Text-to-video
20
+ video = pipe(
21
+ prompt="一名宇航员身穿太空服,面朝镜头骑着一匹机械马在火星表面驰骋。红色的荒凉地表延伸至远方,点缀着巨大的陨石坑和奇特的岩石结构。机械马的步伐稳健,扬起微弱的尘埃,展现出未来科技与原始探索的完美结合。宇航员手持操控装置,目光坚定,仿佛正在开辟人类的新疆域。背景是深邃的宇宙和蔚蓝的地球,画面既科幻又充满希望,让人不禁畅想未来的星际生活。",
22
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
23
+ seed=0, tiled=True,
24
+ )
25
+ if dist.get_rank() == 0:
26
+ save_video(video, "video1.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/LongCat-Video.py ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from diffsynth.utils.data import save_video, VideoData
3
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
4
+
5
+
6
+ pipe = WanVideoPipeline.from_pretrained(
7
+ torch_dtype=torch.bfloat16,
8
+ device="cuda",
9
+ model_configs=[
10
+ ModelConfig(model_id="meituan-longcat/LongCat-Video", origin_file_pattern="dit/diffusion_pytorch_model*.safetensors"),
11
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
12
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="Wan2.1_VAE.pth"),
13
+ ],
14
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
15
+ )
16
+
17
+ # Text-to-video
18
+ video = pipe(
19
+ prompt="In a realistic photography style, a white boy around seven or eight years old sits on a park bench, wearing a light blue T-shirt, denim shorts, and white sneakers. He holds an ice cream cone with vanilla and chocolate flavors, and beside him is a medium-sized golden Labrador. Smiling, the boy offers the ice cream to the dog, who eagerly licks it with its tongue. The sun is shining brightly, and the background features a green lawn and several tall trees, creating a warm and loving scene.",
20
+ negative_prompt="Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards",
21
+ seed=0, tiled=True, num_frames=93,
22
+ cfg_scale=2, sigma_shift=1,
23
+ )
24
+ save_video(video, "video_1_LongCat-Video.mp4", fps=15, quality=5)
25
+
26
+ # Video-continuation (The number of frames in `longcat_video` should be 4n+1.)
27
+ longcat_video = video[-17:]
28
+ video = pipe(
29
+ prompt="In a realistic photography style, a white boy around seven or eight years old sits on a park bench, wearing a light blue T-shirt, denim shorts, and white sneakers. He holds an ice cream cone with vanilla and chocolate flavors, and beside him is a medium-sized golden Labrador. Smiling, the boy offers the ice cream to the dog, who eagerly licks it with its tongue. The sun is shining brightly, and the background features a green lawn and several tall trees, creating a warm and loving scene.",
30
+ negative_prompt="Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards",
31
+ seed=1, tiled=True, num_frames=93,
32
+ cfg_scale=2, sigma_shift=1,
33
+ longcat_video=longcat_video,
34
+ )
35
+ save_video(video, "video_2_LongCat-Video.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Video-As-Prompt-Wan2.1-14B.py ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ import PIL
3
+ from PIL import Image
4
+ from diffsynth.utils.data import save_video, VideoData
5
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
6
+ from modelscope import dataset_snapshot_download
7
+ from typing import List
8
+
9
+
10
+ pipe = WanVideoPipeline.from_pretrained(
11
+ torch_dtype=torch.bfloat16,
12
+ device="cuda",
13
+ model_configs=[
14
+ ModelConfig(model_id="ByteDance/Video-As-Prompt-Wan2.1-14B", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
15
+ ModelConfig(model_id="Wan-AI/Wan2.1-I2V-14B-720P", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
16
+ ModelConfig(model_id="Wan-AI/Wan2.1-I2V-14B-720P", origin_file_pattern="Wan2.1_VAE.pth"),
17
+ ModelConfig(model_id="Wan-AI/Wan2.1-I2V-14B-720P", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
18
+ ],
19
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
20
+ )
21
+
22
+ dataset_snapshot_download("DiffSynth-Studio/example_video_dataset", allow_file_pattern="wanvap/*", local_dir="data/example_video_dataset")
23
+ ref_video_path = 'data/example_video_dataset/wanvap/vap_ref.mp4'
24
+ target_image_path = 'data/example_video_dataset/wanvap/input_image.jpg'
25
+
26
+ def select_frames(video_frames, num):
27
+ idx = torch.linspace(0, len(video_frames) - 1, num).long().tolist()
28
+ return [video_frames[i] for i in idx]
29
+
30
+ image = Image.open(target_image_path).convert("RGB")
31
+ ref_video = VideoData(ref_video_path, height=480, width=832)
32
+ ref_frames = select_frames(ref_video, num=49)
33
+
34
+ vap_prompt = "A man stands with his back to the camera on a dirt path overlooking sun-drenched, rolling green tea plantations. He wears a blue and green plaid shirt, dark pants, and white shoes. As he turns to face the camera and spreads his arms, a brief, magical burst of sparkling golden light particles envelops him. Through this shimmer, he seamlessly transforms into a Labubu toy character. His head morphs into the iconic large, furry-eared head of the toy, featuring a wide grin with pointed teeth and red cheek markings. The character retains the man's original plaid shirt and clothing, which now fit its stylized, cartoonish body. The camera remains static throughout the transformation, positioned low among the tea bushes, maintaining a consistent view of the subject and the expansive scenery."
35
+ prompt = "A young woman with curly hair, wearing a green hijab and a floral dress, plays a violin in front of a vintage green car on a tree-lined street. She executes a swift counter-clockwise turn to face the camera. During the turn, a brilliant shower of golden, sparkling particles erupts and momentarily obscures her figure. As the particles fade, she is revealed to have seamlessly transformed into a Labubu toy character. This new figure, now with the toy's signature large ears, big eyes, and toothy grin, maintains the original pose and continues playing the violin. The character's clothing—the green hijab, floral dress, and black overcoat—remains identical to the woman's. Throughout this transition, the camera stays static, and the street-side environment remains completely consistent."
36
+ negative_prompt = "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
37
+
38
+ video = pipe(
39
+ prompt=prompt,
40
+ negative_prompt=negative_prompt,
41
+ input_image=image,
42
+ seed=42, tiled=True,
43
+ height=480, width=832,
44
+ num_frames=49,
45
+ vap_video=ref_frames,
46
+ vap_prompt=vap_prompt,
47
+ negative_vap_prompt=negative_prompt,
48
+ )
49
+ save_video(video, "video_Video-As-Prompt-Wan2.1-14B.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan-Dancer-14B-global.py ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="global_model.safetensors"),
13
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+ dataset_snapshot_download(
20
+ "DiffSynth-Studio/diffsynth_example_dataset",
21
+ local_dir="data/diffsynth_example_dataset",
22
+ allow_file_pattern="wanvideo/Wan-Dancer-14B-global/*"
23
+ )
24
+ # This is a specialized model with the following constraints on its input parameters:
25
+ # * The model outputs a sequence of keyframes rather than a video; therefore, `framewise_decoding=True` must be set.
26
+ # * When the number of keyframes is $n$, `num_frames` = 4 * (n - 1) + 1.
27
+ # * Reducing `height`, `width`, `num_frames`, or `num_inference_steps` may lead to severe artifacts or generation failure.
28
+ # * The audio file specified by `wantodance_music_path` must match the video duration, calculated as (`num_frames` / 7.5) seconds.
29
+ # * The width and height of `wantodance_reference_image` must be multiples of 16.
30
+ # * `wantodance_fps` is configurable, but since the model appears to have been trained exclusively at 7.5 FPS, setting it to other values is not recommended.
31
+ # * The first frame of `wantodance_keyframes` is the `wantodance_reference_image`, while all subsequent frames are solid black.
32
+ # * `wantodance_keyframes_mask` indicates the positions of valid frames within `wantodance_keyframes`.
33
+ wantodance_keyframes = VideoData("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-global/keyframes.mp4")
34
+ wantodance_keyframes = [wantodance_keyframes[i] for i in range(149)]
35
+ video = pipe(
36
+ prompt="一个人正在跳舞,舞蹈种类是韩舞。帧率是7.5000",
37
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
38
+ seed=0, tiled=False,
39
+ height=1280, width=720, num_frames=149,
40
+ num_inference_steps=48,
41
+ wantodance_music_path="data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-global/music.WAV",
42
+ wantodance_reference_image=Image.open("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-global/refimage.jpg"),
43
+ wantodance_fps=7.5,
44
+ wantodance_keyframes=wantodance_keyframes,
45
+ wantodance_keyframes_mask=[1] + [0] * 148,
46
+ framewise_decoding=True,
47
+ )
48
+ save_video(video, "video_Wan-Dancer-14B-global.mp4", fps=7.5, quality=5)
examples/wanvideo/model_inference/Wan-Dancer-14B-local.py ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch, os
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="local_model.safetensors"),
13
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+ dataset_snapshot_download(
20
+ "DiffSynth-Studio/diffsynth_example_dataset",
21
+ local_dir="data/diffsynth_example_dataset",
22
+ allow_file_pattern="wanvideo/Wan-Dancer-14B-local/*"
23
+ )
24
+ # This is a specialized model with the following constraints on its input parameters:
25
+ # * The model renders and outputs video based on a sequence of keyframes; therefore, `wantodance_keyframes` must be provided correctly.
26
+ # * If you need to generate a long video, please generate it in segments, and ensure that `wantodance_music_path`, `wantodance_keyframes`, and `wantodance_keyframes_mask` are properly split accordingly.
27
+ # * The audio file specified by `wantodance_music_path` must match the video duration, calculated as (`num_frames` / 30) seconds.
28
+ # * The width and height of `wantodance_reference_image` must be multiples of 16.
29
+ # * `wantodance_fps` is configurable, but since the model appears to have been trained exclusively at 30 FPS, setting it to other values is not recommended.
30
+ # * In `wantodance_keyframes`, frames that are not keyframes should be solid black.
31
+ # * `wantodance_keyframes_mask` indicates the positions of valid frames within `wantodance_keyframes`.
32
+ wantodance_keyframes = VideoData("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-local/keyframes.mp4")
33
+ wantodance_keyframes = [wantodance_keyframes[i] for i in range(149)]
34
+ video = pipe(
35
+ prompt="一个人正在跳舞,舞蹈种类是古典舞,图像清晰程度高,人物动作平均幅度中等,人物动作最大幅度中等。, 帧率是30fps。",
36
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
37
+ seed=0, tiled=True,
38
+ height=1280, width=720, num_frames=149,
39
+ num_inference_steps=24,
40
+ wantodance_music_path="data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-local/music.wav",
41
+ wantodance_reference_image=Image.open("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-local/refimage.jpg"),
42
+ wantodance_fps=30,
43
+ wantodance_keyframes=wantodance_keyframes,
44
+ wantodance_keyframes_mask=[1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
45
+ 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
46
+ 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
47
+ 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
48
+ 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
49
+ 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
50
+ 1],
51
+ )
52
+ save_video(video, "video_Wan-Dancer-14B-local.mp4", fps=30, quality=5)
examples/wanvideo/model_inference/Wan2.1-1.3b-speedcontrol-v1.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+
6
+
7
+ pipe = WanVideoPipeline.from_pretrained(
8
+ torch_dtype=torch.bfloat16,
9
+ device="cuda",
10
+ model_configs=[
11
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
12
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
13
+ ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
14
+ ModelConfig(model_id="DiffSynth-Studio/Wan2.1-1.3b-speedcontrol-v1", origin_file_pattern="model.safetensors"),
15
+ ],
16
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
17
+ )
18
+
19
+ # Text-to-video
20
+ video = pipe(
21
+ prompt="纪实摄影风格画面,一只活泼的小狗在绿茵茵的草地上迅速奔跑。小狗毛色棕黄,两只耳朵立起,神情专注而欢快。阳光洒在它身上,使得毛发看上去格外柔软而闪亮。背景是一片开阔的草地,偶尔点缀着几朵野花,远处隐约可见蓝天和几片白云。透视感鲜明,捕捉小狗奔跑时的动感和四周草地的生机。中景侧面移动视角。",
22
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
23
+ seed=1, tiled=True,
24
+ motion_bucket_id=0
25
+ )
26
+ save_video(video, "video_slow_Wan2.1-1.3b-speedcontrol-v1.mp4", fps=15, quality=5)
27
+
28
+ video = pipe(
29
+ prompt="纪实摄影风格画面,一只活泼的小狗在绿茵茵的草地上迅速奔跑。小狗毛色棕黄,两只耳朵立起,神情专注而欢快。阳光洒在它身上,使得毛发看上去格外柔软而闪亮。背景是一片开阔的草地,偶尔点缀着几朵野花,远处隐约可见蓝天和几片白云。透视感鲜明,捕捉小狗奔跑时的动感和四周草地的生机。中景侧面移动视角。",
30
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
31
+ seed=1, tiled=True,
32
+ motion_bucket_id=100
33
+ )
34
+ save_video(video, "video_fast_Wan2.1-1.3b-speedcontrol-v1.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan2.1-FLF2V-14B-720P.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+
20
+ dataset_snapshot_download(
21
+ dataset_id="DiffSynth-Studio/examples_in_diffsynth",
22
+ local_dir="./",
23
+ allow_file_pattern=["data/examples/wan/first_frame.jpeg", "data/examples/wan/last_frame.jpeg"]
24
+ )
25
+
26
+ # First and last frame to video
27
+ video = pipe(
28
+ prompt="写实风格,一个女生手持枯萎的花站在花园中,镜头逐渐拉远,记录下花园的全貌。",
29
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
30
+ input_image=Image.open("data/examples/wan/first_frame.jpeg").resize((960, 960)),
31
+ end_image=Image.open("data/examples/wan/last_frame.jpeg").resize((960, 960)),
32
+ seed=0, tiled=True,
33
+ height=960, width=960, num_frames=33,
34
+ sigma_shift=16,
35
+ )
36
+ save_video(video, "video_Wan2.1-FLF2V-14B-720P.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-Control.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+
20
+ dataset_snapshot_download(
21
+ dataset_id="DiffSynth-Studio/examples_in_diffsynth",
22
+ local_dir="./",
23
+ allow_file_pattern=f"data/examples/wan/control_video.mp4"
24
+ )
25
+
26
+ # Control video
27
+ control_video = VideoData("data/examples/wan/control_video.mp4", height=832, width=576)
28
+ video = pipe(
29
+ prompt="扁平风格动漫,一位长发少女优雅起舞。她五官精致,大眼睛明亮有神,黑色长发柔顺光泽。身穿淡蓝色T恤和深蓝色牛仔短裤。背景是粉色。",
30
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
31
+ control_video=control_video, height=832, width=576, num_frames=49,
32
+ seed=1, tiled=True
33
+ )
34
+ save_video(video, "video_Wan2.1-Fun-1.3B-Control.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-InP.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+
20
+ dataset_snapshot_download(
21
+ dataset_id="DiffSynth-Studio/examples_in_diffsynth",
22
+ local_dir="./",
23
+ allow_file_pattern=f"data/examples/wan/input_image.jpg"
24
+ )
25
+ image = Image.open("data/examples/wan/input_image.jpg")
26
+
27
+ # First and last frame to video
28
+ video = pipe(
29
+ prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
30
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
31
+ input_image=image,
32
+ seed=0, tiled=True
33
+ # You can input `end_image=xxx` to control the last frame of the video.
34
+ # The model will automatically generate the dynamic content between `input_image` and `end_image`.
35
+ )
36
+ save_video(video, "video_Wan2.1-Fun-1.3B-InP.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan2.1-Fun-14B-Control.py ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+
20
+ dataset_snapshot_download(
21
+ dataset_id="DiffSynth-Studio/examples_in_diffsynth",
22
+ local_dir="./",
23
+ allow_file_pattern=f"data/examples/wan/control_video.mp4"
24
+ )
25
+
26
+ # Control video
27
+ control_video = VideoData("data/examples/wan/control_video.mp4", height=832, width=576)
28
+ video = pipe(
29
+ prompt="扁平风格动漫,一位长发少女优雅起舞。她五官精致,大眼睛明亮有神,黑色长发柔顺光泽。身穿淡蓝色T恤和深蓝色牛仔短裤。背景是粉色。",
30
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
31
+ control_video=control_video, height=832, width=576, num_frames=49,
32
+ seed=1, tiled=True
33
+ )
34
+ save_video(video, "video_Wan2.1-Fun-14B-Control.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan2.1-Fun-14B-InP.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+
20
+ dataset_snapshot_download(
21
+ dataset_id="DiffSynth-Studio/examples_in_diffsynth",
22
+ local_dir="./",
23
+ allow_file_pattern=f"data/examples/wan/input_image.jpg"
24
+ )
25
+ image = Image.open("data/examples/wan/input_image.jpg")
26
+
27
+ # First and last frame to video
28
+ video = pipe(
29
+ prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
30
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
31
+ input_image=image,
32
+ seed=0, tiled=True
33
+ # You can input `end_image=xxx` to control the last frame of the video.
34
+ # The model will automatically generate the dynamic content between `input_image` and `end_image`.
35
+ )
36
+ save_video(video, "video_Wan2.1-Fun-14B-InP.mp4", fps=15, quality=5)
examples/wanvideo/model_inference/Wan2.1-Fun-V1.1-1.3B-Control-Camera.py ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from PIL import Image
3
+ from diffsynth.utils.data import save_video, VideoData
4
+ from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
5
+ from modelscope import dataset_snapshot_download
6
+
7
+
8
+ pipe = WanVideoPipeline.from_pretrained(
9
+ torch_dtype=torch.bfloat16,
10
+ device="cuda",
11
+ model_configs=[
12
+ ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
13
+ ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
14
+ ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="Wan2.1_VAE.pth"),
15
+ ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
16
+ ],
17
+ tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
18
+ )
19
+
20
+
21
+ dataset_snapshot_download(
22
+ dataset_id="DiffSynth-Studio/examples_in_diffsynth",
23
+ local_dir="./",
24
+ allow_file_pattern=f"data/examples/wan/input_image.jpg"
25
+ )
26
+ input_image = Image.open("data/examples/wan/input_image.jpg")
27
+
28
+ video = pipe(
29
+ prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
30
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
31
+ seed=0, tiled=True,
32
+ input_image=input_image,
33
+ camera_control_direction="Left", camera_control_speed=0.01,
34
+ )
35
+ save_video(video, "video_left_Wan2.1-Fun-V1.1-1.3B-Control-Camera.mp4", fps=15, quality=5)
36
+
37
+ video = pipe(
38
+ prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
39
+ negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
40
+ seed=0, tiled=True,
41
+ input_image=input_image,
42
+ camera_control_direction="Up", camera_control_speed=0.01,
43
+ )
44
+ save_video(video, "video_up_Wan2.1-Fun-V1.1-1.3B-Control-Camera.mp4", fps=15, quality=5)