Upload folder using huggingface_hub (part 5)
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +2 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit-2511.py +24 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit.py +21 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen-Poster.py +29 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen.py +29 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-In-Context-Control-Union.py +19 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control-V2.py +37 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control.py +25 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered.py +27 -0
- examples/qwen_image/model_training/validate_lora/Qwen-Image.py +18 -0
- examples/qwen_video_edit/model_inference/Qwen-Video-Edit.py +27 -0
- examples/qwen_video_edit/model_inference_low_vram/Qwen-Video-Edit.py +40 -0
- examples/qwen_video_edit/model_training/full/Qwen-Video-Edit.sh +19 -0
- examples/qwen_video_edit/model_training/full/accelerate_config_zero3.yaml +23 -0
- examples/qwen_video_edit/model_training/lora/Qwen-Video-Edit.sh +22 -0
- examples/qwen_video_edit/model_training/train.py +175 -0
- examples/qwen_video_edit/model_training/validate_full/Qwen-Video-Edit.py +25 -0
- examples/qwen_video_edit/model_training/validate_lora/Qwen-Video-Edit.py +23 -0
- examples/stable_diffusion/model_inference/stable-diffusion-v1-5.py +25 -0
- examples/stable_diffusion/model_inference_low_vram/stable-diffusion-v1-5.py +36 -0
- examples/stable_diffusion/model_training/full/stable-diffusion-v1-5.sh +15 -0
- examples/stable_diffusion/model_training/lora/stable-diffusion-v1-5.sh +17 -0
- examples/stable_diffusion/model_training/special/split_training/stable-diffusion-v1-5.sh +38 -0
- examples/stable_diffusion/model_training/special/split_training/validate.py +26 -0
- examples/stable_diffusion/model_training/train.py +154 -0
- examples/stable_diffusion/model_training/validate_full/stable-diffusion-v1-5.py +27 -0
- examples/stable_diffusion/model_training/validate_lora/stable-diffusion-v1-5.py +26 -0
- examples/stable_diffusion_xl/model_inference/stable-diffusion-xl-base-1.0.py +26 -0
- examples/stable_diffusion_xl/model_inference_low_vram/stable-diffusion-xl-base-1.0.py +37 -0
- examples/stable_diffusion_xl/model_training/full/stable-diffusion-xl-base-1.0.sh +15 -0
- examples/stable_diffusion_xl/model_training/lora/stable-diffusion-xl-base-1.0.sh +18 -0
- examples/stable_diffusion_xl/model_training/special/split_training/stable-diffusion-xl-base-1.0.sh +40 -0
- examples/stable_diffusion_xl/model_training/special/split_training/validate.py +27 -0
- examples/stable_diffusion_xl/model_training/train.py +159 -0
- examples/stable_diffusion_xl/model_training/validate_full/stable-diffusion-xl-base-1.0.py +28 -0
- examples/stable_diffusion_xl/model_training/validate_lora/stable-diffusion-xl-base-1.0.py +27 -0
- examples/wanvideo/README.md +3 -0
- examples/wanvideo/acceleration/Wan2.2-Animate-2-14B-usp.py +117 -0
- examples/wanvideo/acceleration/unified_sequence_parallel.py +26 -0
- examples/wanvideo/model_inference/LongCat-Video.py +35 -0
- examples/wanvideo/model_inference/Video-As-Prompt-Wan2.1-14B.py +49 -0
- examples/wanvideo/model_inference/Wan-Dancer-14B-global.py +48 -0
- examples/wanvideo/model_inference/Wan-Dancer-14B-local.py +52 -0
- examples/wanvideo/model_inference/Wan2.1-1.3b-speedcontrol-v1.py +34 -0
- examples/wanvideo/model_inference/Wan2.1-FLF2V-14B-720P.py +36 -0
- examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-Control.py +34 -0
- examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-InP.py +36 -0
- examples/wanvideo/model_inference/Wan2.1-Fun-14B-Control.py +34 -0
- examples/wanvideo/model_inference/Wan2.1-Fun-14B-InP.py +36 -0
- examples/wanvideo/model_inference/Wan2.1-Fun-V1.1-1.3B-Control-Camera.py +44 -0
.gitattributes
CHANGED
|
@@ -34,3 +34,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
.github/workflows/logo.gif filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
.github/workflows/logo.gif filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
models/train/MiniMax-H3-Ref2VA-CineDance-filtered-multishot-native-partial/wandb_log/wandb/run-20260916_162111-qrso8o89/run-qrso8o89.wandb filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
models/train/MiniMax-H3-Ref2VA-CineDance-multishot-full-v2/wandb_log/wandb/run-20260831_033451-73yd8zqv/run-73yd8zqv.wandb filter=lfs diff=lfs merge=lfs -text
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit-2511.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 4 |
+
|
| 5 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.bfloat16,
|
| 7 |
+
device="cuda",
|
| 8 |
+
model_configs=[
|
| 9 |
+
ModelConfig(model_id="Qwen/Qwen-Image-Edit-2511", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 10 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=None,
|
| 14 |
+
processor_config=ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="processor/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Edit-2511_lora/epoch-4.safetensors")
|
| 17 |
+
|
| 18 |
+
prompt = "Change the color of the dress in Figure 1 to the color shown in Figure 2."
|
| 19 |
+
images = [
|
| 20 |
+
Image.open("data/example_image_dataset/edit/image1.jpg").resize((1024, 1024)),
|
| 21 |
+
Image.open("data/example_image_dataset/edit/image_color.jpg").resize((1024, 1024)),
|
| 22 |
+
]
|
| 23 |
+
image = pipe(prompt, edit_image=images, seed=123, num_inference_steps=40, height=1024, width=1024, zero_cond_t=True)
|
| 24 |
+
image.save("image.jpg")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-Edit.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 4 |
+
|
| 5 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.bfloat16,
|
| 7 |
+
device="cuda",
|
| 8 |
+
model_configs=[
|
| 9 |
+
ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 10 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=None,
|
| 14 |
+
processor_config=ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="processor/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Edit_lora/epoch-4.safetensors")
|
| 17 |
+
|
| 18 |
+
prompt = "将裙子改为粉色"
|
| 19 |
+
image = Image.open("data/example_image_dataset/edit/image1.jpg").resize((1024, 1024))
|
| 20 |
+
image = pipe(prompt, edit_image=image, seed=0, num_inference_steps=40, height=1024, width=1024)
|
| 21 |
+
image.save(f"image.jpg")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen-Poster.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 2 |
+
import torch
|
| 3 |
+
from PIL import Image
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 7 |
+
torch_dtype=torch.bfloat16,
|
| 8 |
+
device="cuda",
|
| 9 |
+
model_configs=[
|
| 10 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 12 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 13 |
+
],
|
| 14 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-EliGen-Poster_lora/epoch-4.safetensors")
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
entity_prompts = ["A beautiful girl", "sign 'Entity Control'", "shorts", "shirt"]
|
| 20 |
+
global_prompt = "A beautiful girl wearing shirt and shorts in the street, holding a sign 'Entity Control'"
|
| 21 |
+
masks = [Image.open(f"data/example_image_dataset/eligen/{i}.png").convert('RGB') for i in range(len(entity_prompts))]
|
| 22 |
+
|
| 23 |
+
image = pipe(global_prompt,
|
| 24 |
+
seed=0,
|
| 25 |
+
height=1024,
|
| 26 |
+
width=1024,
|
| 27 |
+
eligen_entity_prompts=entity_prompts,
|
| 28 |
+
eligen_entity_masks=masks)
|
| 29 |
+
image.save("Qwen-Image-EliGen-Poster.jpg")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-EliGen.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 2 |
+
import torch
|
| 3 |
+
from PIL import Image
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 7 |
+
torch_dtype=torch.bfloat16,
|
| 8 |
+
device="cuda",
|
| 9 |
+
model_configs=[
|
| 10 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 12 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 13 |
+
],
|
| 14 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-EliGen_lora/epoch-4.safetensors")
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
entity_prompts = ["A beautiful girl", "sign 'Entity Control'", "shorts", "shirt"]
|
| 20 |
+
global_prompt = "A beautiful girl wearing shirt and shorts in the street, holding a sign 'Entity Control'"
|
| 21 |
+
masks = [Image.open(f"data/example_image_dataset/eligen/{i}.png").convert('RGB') for i in range(len(entity_prompts))]
|
| 22 |
+
|
| 23 |
+
image = pipe(global_prompt,
|
| 24 |
+
seed=0,
|
| 25 |
+
height=1024,
|
| 26 |
+
width=1024,
|
| 27 |
+
eligen_entity_prompts=entity_prompts,
|
| 28 |
+
eligen_entity_masks=masks)
|
| 29 |
+
image.save("Qwen-Image_EliGen.jpg")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-In-Context-Control-Union.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from PIL import Image
|
| 2 |
+
import torch
|
| 3 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 4 |
+
|
| 5 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.bfloat16,
|
| 7 |
+
device="cuda",
|
| 8 |
+
model_configs=[
|
| 9 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 10 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 14 |
+
)
|
| 15 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-In-Context-Control-Union_lora/epoch-4.safetensors")
|
| 16 |
+
image = Image.open("data/example_image_dataset/canny/image_1.jpg").resize((1024, 1024))
|
| 17 |
+
prompt = "Context_Control. a dog"
|
| 18 |
+
image = pipe(prompt=prompt, seed=0, context_image=image, height=1024, width=1024)
|
| 19 |
+
image.save("image_context.jpg")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control-V2.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 2 |
+
from modelscope import dataset_snapshot_download
|
| 3 |
+
from PIL import Image
|
| 4 |
+
import torch
|
| 5 |
+
|
| 6 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 7 |
+
torch_dtype=torch.bfloat16,
|
| 8 |
+
device="cuda",
|
| 9 |
+
model_configs=[
|
| 10 |
+
ModelConfig(model_id="DiffSynth-Studio/Qwen-Image-Layered-Control", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 12 |
+
ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 13 |
+
],
|
| 14 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Layered-Control-V2_lora/epoch-4.safetensors")
|
| 17 |
+
|
| 18 |
+
prompt = "Text 'APRIL'"
|
| 19 |
+
input_image = Image.open("data/example_image_dataset/layer_v2/image_1.png").convert("RGBA").resize((1024, 1024))
|
| 20 |
+
image = pipe(
|
| 21 |
+
prompt, seed=0,
|
| 22 |
+
height=1024, width=1024,
|
| 23 |
+
layer_input_image=input_image, layer_num=0,
|
| 24 |
+
num_inference_steps=10, cfg_scale=4,
|
| 25 |
+
)
|
| 26 |
+
image[0].save("image_prompt.png")
|
| 27 |
+
|
| 28 |
+
mask_image = Image.open("data/example_image_dataset/layer_v2/mask_2.png").convert("RGBA").resize((1024, 1024))
|
| 29 |
+
input_image = Image.open("data/example_image_dataset/layer_v2/image_2.png").convert("RGBA").resize((1024, 1024))
|
| 30 |
+
image = pipe(
|
| 31 |
+
prompt, seed=0,
|
| 32 |
+
height=1024, width=1024,
|
| 33 |
+
layer_input_image=input_image, layer_num=0,
|
| 34 |
+
context_image=mask_image,
|
| 35 |
+
num_inference_steps=10, cfg_scale=1.0,
|
| 36 |
+
)
|
| 37 |
+
image[0].save("image_mask.png")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered-Control.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 2 |
+
from diffsynth import load_state_dict
|
| 3 |
+
from PIL import Image
|
| 4 |
+
import torch
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 8 |
+
torch_dtype=torch.bfloat16,
|
| 9 |
+
device="cuda",
|
| 10 |
+
model_configs=[
|
| 11 |
+
ModelConfig(model_id="DiffSynth-Studio/Qwen-Image-Layered-Control", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 12 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 14 |
+
],
|
| 15 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 16 |
+
)
|
| 17 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Layered-Control_lora/epoch-4.safetensors")
|
| 18 |
+
prompt = "Text 'HELLO' and 'Have a great day'"
|
| 19 |
+
input_image = Image.open("data/example_image_dataset/layer/image.png").convert("RGBA").resize((864, 480))
|
| 20 |
+
images = pipe(
|
| 21 |
+
prompt, seed=0,
|
| 22 |
+
height=480, width=864,
|
| 23 |
+
layer_input_image=input_image, layer_num=0,
|
| 24 |
+
)
|
| 25 |
+
images[0].save("image.png")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image-Layered.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 2 |
+
from diffsynth import load_state_dict
|
| 3 |
+
from PIL import Image
|
| 4 |
+
import torch
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 8 |
+
torch_dtype=torch.bfloat16,
|
| 9 |
+
device="cuda",
|
| 10 |
+
model_configs=[
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 12 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="Qwen/Qwen-Image-Layered", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 14 |
+
],
|
| 15 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 16 |
+
)
|
| 17 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Layered_lora/epoch-4.safetensors")
|
| 18 |
+
prompt = "a poster"
|
| 19 |
+
input_image = Image.open("data/example_image_dataset/layer/image.png").convert("RGBA").resize((864, 480))
|
| 20 |
+
images = pipe(
|
| 21 |
+
prompt, seed=0,
|
| 22 |
+
height=480, width=864,
|
| 23 |
+
layer_input_image=input_image, layer_num=3,
|
| 24 |
+
)
|
| 25 |
+
for i, image in enumerate(images):
|
| 26 |
+
if i == 0: continue # The first image is the input image.
|
| 27 |
+
image.save(f"image_{i}.png")
|
examples/qwen_image/model_training/validate_lora/Qwen-Image.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
|
| 2 |
+
import torch
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
pipe = QwenImagePipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.bfloat16,
|
| 7 |
+
device="cuda",
|
| 8 |
+
model_configs=[
|
| 9 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 10 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
|
| 14 |
+
)
|
| 15 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Image_lora/epoch-4.safetensors")
|
| 16 |
+
prompt = "a dog"
|
| 17 |
+
image = pipe(prompt, seed=0)
|
| 18 |
+
image.save("image.jpg")
|
examples/qwen_video_edit/model_inference/Qwen-Video-Edit.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from modelscope import dataset_snapshot_download
|
| 3 |
+
from diffsynth.core import ModelConfig
|
| 4 |
+
from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
|
| 5 |
+
from diffsynth.utils.data import VideoData, save_video
|
| 6 |
+
|
| 7 |
+
dataset_snapshot_download(
|
| 8 |
+
"DiffSynth-Studio/diffsynth_example_dataset",
|
| 9 |
+
local_dir="data/diffsynth_example_dataset",
|
| 10 |
+
allow_file_pattern="qwen_video_edit/Qwen-Video-Edit/*"
|
| 11 |
+
)
|
| 12 |
+
|
| 13 |
+
edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
|
| 14 |
+
prompts = [
|
| 15 |
+
"Transform the video into Japanese anime style",
|
| 16 |
+
]
|
| 17 |
+
pipe = QwenVideoEditPipeline.from_pretrained(
|
| 18 |
+
torch_dtype=torch.bfloat16,
|
| 19 |
+
device="cuda",
|
| 20 |
+
model_configs=[
|
| 21 |
+
ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors"),
|
| 22 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 23 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 24 |
+
],
|
| 25 |
+
)
|
| 26 |
+
video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
|
| 27 |
+
save_video(video, "video_Qwen-Video-Edit.mp4", fps=16)
|
examples/qwen_video_edit/model_inference_low_vram/Qwen-Video-Edit.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from modelscope import dataset_snapshot_download
|
| 3 |
+
from diffsynth.core import ModelConfig
|
| 4 |
+
from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
|
| 5 |
+
from diffsynth.utils.data import VideoData, save_video
|
| 6 |
+
|
| 7 |
+
vram_config = {
|
| 8 |
+
"offload_dtype": torch.bfloat16,
|
| 9 |
+
"offload_device": "cpu",
|
| 10 |
+
"onload_dtype": torch.bfloat16,
|
| 11 |
+
"onload_device": "cpu",
|
| 12 |
+
"preparing_dtype": torch.bfloat16,
|
| 13 |
+
"preparing_device": "cuda",
|
| 14 |
+
"computation_dtype": torch.bfloat16,
|
| 15 |
+
"computation_device": "cuda",
|
| 16 |
+
}
|
| 17 |
+
|
| 18 |
+
dataset_snapshot_download(
|
| 19 |
+
"DiffSynth-Studio/diffsynth_example_dataset",
|
| 20 |
+
local_dir="data/diffsynth_example_dataset",
|
| 21 |
+
allow_file_pattern="qwen_video_edit/Qwen-Video-Edit/*"
|
| 22 |
+
)
|
| 23 |
+
|
| 24 |
+
edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
|
| 25 |
+
|
| 26 |
+
prompts = [
|
| 27 |
+
"Transform the video into Japanese anime style",
|
| 28 |
+
]
|
| 29 |
+
pipe = QwenVideoEditPipeline.from_pretrained(
|
| 30 |
+
torch_dtype=torch.bfloat16,
|
| 31 |
+
device="cuda",
|
| 32 |
+
model_configs=[
|
| 33 |
+
ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors", **vram_config),
|
| 34 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors", **vram_config),
|
| 35 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth", **vram_config),
|
| 36 |
+
],
|
| 37 |
+
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
|
| 38 |
+
)
|
| 39 |
+
video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
|
| 40 |
+
save_video(video, "video_Qwen-Video-Edit.mp4", fps=16)
|
examples/qwen_video_edit/model_training/full/Qwen-Video-Edit.sh
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "qwen_video_edit/Qwen-Video-Edit/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
accelerate launch examples/qwen_video_edit/model_training/train.py \
|
| 4 |
+
--dataset_base_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit \
|
| 5 |
+
--dataset_metadata_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/metadata.json \
|
| 6 |
+
--data_file_keys "video,input_video" \
|
| 7 |
+
--height 640 \
|
| 8 |
+
--width 384 \
|
| 9 |
+
--num_frames 45 \
|
| 10 |
+
--dataset_repeat 50 \
|
| 11 |
+
--model_id_with_origin_paths "yunpeng1998/Qwen-Video-Edit:360P/step-30000.safetensors,Qwen/Qwen-Image:text_encoder/model*.safetensors,Wan-AI/Wan2.1-T2V-1.3B:Wan2.1_VAE.pth" \
|
| 12 |
+
--learning_rate 1e-5 \
|
| 13 |
+
--num_epochs 2 \
|
| 14 |
+
--remove_prefix_in_ckpt "pipe.dit." \
|
| 15 |
+
--output_path "./models/train/Qwen-Video-Edit_full" \
|
| 16 |
+
--trainable_models "dit" \
|
| 17 |
+
--use_gradient_checkpointing \
|
| 18 |
+
--zero_cond_t \
|
| 19 |
+
--find_unused_parameters
|
examples/qwen_video_edit/model_training/full/accelerate_config_zero3.yaml
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
compute_environment: LOCAL_MACHINE
|
| 2 |
+
debug: false
|
| 3 |
+
deepspeed_config:
|
| 4 |
+
gradient_accumulation_steps: 1
|
| 5 |
+
offload_optimizer_device: none
|
| 6 |
+
offload_param_device: none
|
| 7 |
+
zero3_init_flag: true
|
| 8 |
+
zero3_save_16bit_model: true
|
| 9 |
+
zero_stage: 3
|
| 10 |
+
distributed_type: DEEPSPEED
|
| 11 |
+
downcast_bf16: 'no'
|
| 12 |
+
enable_cpu_affinity: false
|
| 13 |
+
machine_rank: 0
|
| 14 |
+
main_training_function: main
|
| 15 |
+
mixed_precision: bf16
|
| 16 |
+
num_machines: 1
|
| 17 |
+
num_processes: 8
|
| 18 |
+
rdzv_backend: static
|
| 19 |
+
same_network: true
|
| 20 |
+
tpu_env: []
|
| 21 |
+
tpu_use_cluster: false
|
| 22 |
+
tpu_use_sudo: false
|
| 23 |
+
use_cpu: false
|
examples/qwen_video_edit/model_training/lora/Qwen-Video-Edit.sh
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "qwen_video_edit/Qwen-Video-Edit/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
accelerate launch examples/qwen_video_edit/model_training/train.py \
|
| 4 |
+
--dataset_base_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit \
|
| 5 |
+
--dataset_metadata_path data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/metadata.json \
|
| 6 |
+
--data_file_keys "video,input_video" \
|
| 7 |
+
--height 640 \
|
| 8 |
+
--width 384 \
|
| 9 |
+
--num_frames 45 \
|
| 10 |
+
--dataset_repeat 50 \
|
| 11 |
+
--model_id_with_origin_paths "yunpeng1998/Qwen-Video-Edit:360P/step-30000.safetensors,Qwen/Qwen-Image:text_encoder/model*.safetensors,Wan-AI/Wan2.1-T2V-1.3B:Wan2.1_VAE.pth" \
|
| 12 |
+
--learning_rate 1e-4 \
|
| 13 |
+
--num_epochs 5 \
|
| 14 |
+
--remove_prefix_in_ckpt "pipe.dit." \
|
| 15 |
+
--output_path "./models/train/Qwen-Video-Edit_lora" \
|
| 16 |
+
--lora_base_model "dit" \
|
| 17 |
+
--lora_target_modules "to_q,to_k,to_v,add_q_proj,add_k_proj,add_v_proj,to_out.0,to_add_out,img_mlp.net.2,img_mod.1,txt_mlp.net.2,txt_mod.1" \
|
| 18 |
+
--lora_rank 32 \
|
| 19 |
+
--use_gradient_checkpointing \
|
| 20 |
+
--zero_cond_t \
|
| 21 |
+
--dataset_num_workers 8 \
|
| 22 |
+
--find_unused_parameters
|
examples/qwen_video_edit/model_training/train.py
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch, os, argparse, accelerate, warnings
|
| 2 |
+
from diffsynth.core import UnifiedDataset, ModelConfig
|
| 3 |
+
from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
|
| 4 |
+
from diffsynth.diffusion import *
|
| 5 |
+
from diffsynth.core.data.operators import *
|
| 6 |
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
class QwenVideoEditTrainingModule(DiffusionTrainingModule):
|
| 10 |
+
def __init__(
|
| 11 |
+
self,
|
| 12 |
+
model_paths=None, model_id_with_origin_paths=None,
|
| 13 |
+
tokenizer_path=None, processor_path=None,
|
| 14 |
+
trainable_models=None,
|
| 15 |
+
lora_base_model=None, lora_target_modules="", lora_rank=32, lora_checkpoint=None,
|
| 16 |
+
preset_lora_path=None, preset_lora_model=None,
|
| 17 |
+
use_gradient_checkpointing=True,
|
| 18 |
+
use_gradient_checkpointing_offload=False,
|
| 19 |
+
extra_inputs=None,
|
| 20 |
+
fp8_models=None,
|
| 21 |
+
offload_models=None,
|
| 22 |
+
quant_options=None,
|
| 23 |
+
resume_from_checkpoint=None, remove_prefix_in_ckpt=None,
|
| 24 |
+
device="cpu",
|
| 25 |
+
task="sft",
|
| 26 |
+
zero_cond_t=False,
|
| 27 |
+
max_timestep_boundary=1.0,
|
| 28 |
+
min_timestep_boundary=0.0,
|
| 29 |
+
):
|
| 30 |
+
super().__init__()
|
| 31 |
+
# Load models
|
| 32 |
+
model_configs = self.parse_model_configs(model_paths, model_id_with_origin_paths, fp8_models=fp8_models, offload_models=offload_models, quant_options=quant_options, device=device)
|
| 33 |
+
tokenizer_config = ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/") if tokenizer_path is None else ModelConfig(tokenizer_path)
|
| 34 |
+
processor_config = ModelConfig(model_id="Qwen/Qwen-Image-Edit", origin_file_pattern="processor/") if processor_path is None else ModelConfig(processor_path)
|
| 35 |
+
self.pipe = QwenVideoEditPipeline.from_pretrained(torch_dtype=torch.bfloat16, device=device, model_configs=model_configs, tokenizer_config=tokenizer_config, processor_config=processor_config)
|
| 36 |
+
self.pipe = self.split_pipeline_units(task, self.pipe, trainable_models, lora_base_model)
|
| 37 |
+
self.resume_from_checkpoint(resume_from_checkpoint, remove_prefix_in_ckpt)
|
| 38 |
+
|
| 39 |
+
# Training mode
|
| 40 |
+
self.switch_pipe_to_training_mode(
|
| 41 |
+
self.pipe, trainable_models,
|
| 42 |
+
lora_base_model, lora_target_modules, lora_rank, lora_checkpoint,
|
| 43 |
+
preset_lora_path, preset_lora_model,
|
| 44 |
+
task=task,
|
| 45 |
+
)
|
| 46 |
+
|
| 47 |
+
# Store other configs
|
| 48 |
+
self.use_gradient_checkpointing = use_gradient_checkpointing
|
| 49 |
+
self.use_gradient_checkpointing_offload = use_gradient_checkpointing_offload
|
| 50 |
+
self.extra_inputs = extra_inputs.split(",") if extra_inputs is not None else []
|
| 51 |
+
self.fp8_models = fp8_models
|
| 52 |
+
self.task = task
|
| 53 |
+
self.zero_cond_t = zero_cond_t
|
| 54 |
+
self.max_timestep_boundary = max_timestep_boundary
|
| 55 |
+
self.min_timestep_boundary = min_timestep_boundary
|
| 56 |
+
self.task_to_loss = {
|
| 57 |
+
"sft:data_process": lambda pipe, *args: args,
|
| 58 |
+
"sft": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
|
| 59 |
+
"sft:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
|
| 60 |
+
}
|
| 61 |
+
|
| 62 |
+
def get_pipeline_inputs(self, data):
|
| 63 |
+
inputs_posi = {"prompt": data["prompt"]}
|
| 64 |
+
inputs_nega = {"negative_prompt": ""}
|
| 65 |
+
input_video = data["input_video"]
|
| 66 |
+
num_frames = len(input_video)
|
| 67 |
+
inputs_shared = {
|
| 68 |
+
"edit_video": input_video,
|
| 69 |
+
"input_video": data["video"],
|
| 70 |
+
"chunk_id": 0,
|
| 71 |
+
"num_frames": num_frames,
|
| 72 |
+
"height": input_video[0].size[1],
|
| 73 |
+
"width": input_video[0].size[0],
|
| 74 |
+
"tiled": False,
|
| 75 |
+
"tile_size": (30, 52),
|
| 76 |
+
"tile_stride": (15, 26),
|
| 77 |
+
"cfg_scale": 1,
|
| 78 |
+
"rand_device": self.pipe.device,
|
| 79 |
+
"use_gradient_checkpointing": self.use_gradient_checkpointing,
|
| 80 |
+
"use_gradient_checkpointing_offload": self.use_gradient_checkpointing_offload,
|
| 81 |
+
"zero_cond_t": self.zero_cond_t,
|
| 82 |
+
"max_timestep_boundary": self.max_timestep_boundary,
|
| 83 |
+
"min_timestep_boundary": self.min_timestep_boundary,
|
| 84 |
+
}
|
| 85 |
+
inputs_shared = self.parse_extra_inputs(data, self.extra_inputs, inputs_shared)
|
| 86 |
+
return inputs_shared, inputs_posi, inputs_nega
|
| 87 |
+
|
| 88 |
+
def forward(self, data, inputs=None):
|
| 89 |
+
if inputs is None: inputs = self.get_pipeline_inputs(data)
|
| 90 |
+
inputs = self.transfer_data_to_device(inputs, self.pipe.device, self.pipe.torch_dtype)
|
| 91 |
+
for unit in self.pipe.units:
|
| 92 |
+
inputs = self.pipe.unit_runner(unit, self.pipe, *inputs)
|
| 93 |
+
loss = self.task_to_loss[self.task](self.pipe, *inputs)
|
| 94 |
+
return loss
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
def qwen_video_edit_parser():
|
| 98 |
+
parser = argparse.ArgumentParser(description="Simple example of a training script.")
|
| 99 |
+
parser = add_general_config(parser)
|
| 100 |
+
parser = add_video_size_config(parser)
|
| 101 |
+
parser.add_argument("--tokenizer_path", type=str, default=None, help="Path to tokenizer.")
|
| 102 |
+
parser.add_argument("--processor_path", type=str, default=None, help="Path to the processor. If provided, the processor will be used for image editing.")
|
| 103 |
+
parser.add_argument("--zero_cond_t", default=False, action="store_true", help="A special parameter introduced by Qwen-Image-Edit-2511. Please enable it for this model.")
|
| 104 |
+
parser.add_argument("--max_timestep_boundary", type=float, default=1.0, help="Max timestep boundary.")
|
| 105 |
+
parser.add_argument("--min_timestep_boundary", type=float, default=0.0, help="Min timestep boundary.")
|
| 106 |
+
parser.add_argument("--initialize_model_on_cpu", default=False, action="store_true", help="Whether to initialize models on CPU.")
|
| 107 |
+
return parser
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
if __name__ == "__main__":
|
| 111 |
+
parser = qwen_video_edit_parser()
|
| 112 |
+
args = parser.parse_args()
|
| 113 |
+
accelerator = accelerate.Accelerator(
|
| 114 |
+
gradient_accumulation_steps=args.gradient_accumulation_steps,
|
| 115 |
+
kwargs_handlers=[accelerate.DistributedDataParallelKwargs(find_unused_parameters=args.find_unused_parameters)],
|
| 116 |
+
)
|
| 117 |
+
dataset = UnifiedDataset(
|
| 118 |
+
base_path=args.dataset_base_path,
|
| 119 |
+
metadata_path=args.dataset_metadata_path,
|
| 120 |
+
repeat=args.dataset_repeat,
|
| 121 |
+
data_file_keys=args.data_file_keys.split(","),
|
| 122 |
+
main_data_operator=UnifiedDataset.default_video_operator(
|
| 123 |
+
base_path=args.dataset_base_path,
|
| 124 |
+
max_pixels=args.max_pixels,
|
| 125 |
+
height=args.height,
|
| 126 |
+
width=args.width,
|
| 127 |
+
height_division_factor=16,
|
| 128 |
+
width_division_factor=16,
|
| 129 |
+
num_frames=args.num_frames,
|
| 130 |
+
time_division_factor=4,
|
| 131 |
+
time_division_remainder=1,
|
| 132 |
+
),
|
| 133 |
+
)
|
| 134 |
+
model = QwenVideoEditTrainingModule(
|
| 135 |
+
model_paths=args.model_paths,
|
| 136 |
+
model_id_with_origin_paths=args.model_id_with_origin_paths,
|
| 137 |
+
tokenizer_path=args.tokenizer_path,
|
| 138 |
+
processor_path=args.processor_path,
|
| 139 |
+
trainable_models=args.trainable_models,
|
| 140 |
+
lora_base_model=args.lora_base_model,
|
| 141 |
+
lora_target_modules=args.lora_target_modules,
|
| 142 |
+
lora_rank=args.lora_rank,
|
| 143 |
+
lora_checkpoint=args.lora_checkpoint,
|
| 144 |
+
preset_lora_path=args.preset_lora_path,
|
| 145 |
+
preset_lora_model=args.preset_lora_model,
|
| 146 |
+
use_gradient_checkpointing=args.use_gradient_checkpointing,
|
| 147 |
+
use_gradient_checkpointing_offload=args.use_gradient_checkpointing_offload,
|
| 148 |
+
extra_inputs=args.extra_inputs,
|
| 149 |
+
fp8_models=args.fp8_models,
|
| 150 |
+
offload_models=args.offload_models,
|
| 151 |
+
quant_options=args.quant_options,
|
| 152 |
+
resume_from_checkpoint=args.resume_from_checkpoint,
|
| 153 |
+
remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
|
| 154 |
+
task=args.task,
|
| 155 |
+
device="cpu" if (args.initialize_model_on_cpu or args.enable_model_cpu_offload) else accelerator.device,
|
| 156 |
+
zero_cond_t=args.zero_cond_t,
|
| 157 |
+
max_timestep_boundary=args.max_timestep_boundary,
|
| 158 |
+
min_timestep_boundary=args.min_timestep_boundary,
|
| 159 |
+
)
|
| 160 |
+
model_logger = ModelLogger(
|
| 161 |
+
args.output_path,
|
| 162 |
+
remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
|
| 163 |
+
enable_tensorboard_log=args.enable_tensorboard_log,
|
| 164 |
+
enable_swanlab_log=args.enable_swanlab_log,
|
| 165 |
+
swanlab_project=args.swanlab_project,
|
| 166 |
+
enable_wandb_log=args.enable_wandb_log,
|
| 167 |
+
wandb_project=args.wandb_project,
|
| 168 |
+
enable_csv_log=args.enable_csv_log,
|
| 169 |
+
)
|
| 170 |
+
launcher_map = {
|
| 171 |
+
"sft:data_process": launch_data_process_task,
|
| 172 |
+
"sft": launch_training_task,
|
| 173 |
+
"sft:train": launch_training_task,
|
| 174 |
+
}
|
| 175 |
+
launcher_map[args.task](accelerator, dataset, model, model_logger, args=args)
|
examples/qwen_video_edit/model_training/validate_full/Qwen-Video-Edit.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
|
| 3 |
+
from diffsynth import load_state_dict
|
| 4 |
+
from diffsynth.core import ModelConfig
|
| 5 |
+
from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
|
| 6 |
+
from diffsynth.utils.data import VideoData, save_video
|
| 7 |
+
|
| 8 |
+
edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
|
| 9 |
+
prompts = [
|
| 10 |
+
"Transform the video into Japanese anime style",
|
| 11 |
+
]
|
| 12 |
+
pipe = QwenVideoEditPipeline.from_pretrained(
|
| 13 |
+
torch_dtype=torch.bfloat16,
|
| 14 |
+
device="cuda",
|
| 15 |
+
model_configs=[
|
| 16 |
+
ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors"),
|
| 17 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 18 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 19 |
+
],
|
| 20 |
+
)
|
| 21 |
+
state_dict = load_state_dict("models/train/Qwen-Video-Edit_full/epoch-1.safetensors")
|
| 22 |
+
pipe.dit.load_state_dict(state_dict)
|
| 23 |
+
|
| 24 |
+
video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
|
| 25 |
+
save_video(video, "video_Qwen-Video-Edit-full.mp4", fps=16)
|
examples/qwen_video_edit/model_training/validate_lora/Qwen-Video-Edit.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
|
| 3 |
+
from diffsynth.core import ModelConfig
|
| 4 |
+
from diffsynth.pipelines.qwen_video_edit import QwenVideoEditPipeline
|
| 5 |
+
from diffsynth.utils.data import VideoData, save_video
|
| 6 |
+
|
| 7 |
+
edit_video = VideoData("data/diffsynth_example_dataset/qwen_video_edit/Qwen-Video-Edit/source.mp4")
|
| 8 |
+
prompts = [
|
| 9 |
+
"Transform the video into Japanese anime style",
|
| 10 |
+
]
|
| 11 |
+
pipe = QwenVideoEditPipeline.from_pretrained(
|
| 12 |
+
torch_dtype=torch.bfloat16,
|
| 13 |
+
device="cuda",
|
| 14 |
+
model_configs=[
|
| 15 |
+
ModelConfig(model_id="yunpeng1998/Qwen-Video-Edit", origin_file_pattern="360P/step-30000.safetensors"),
|
| 16 |
+
ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
|
| 17 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 18 |
+
],
|
| 19 |
+
)
|
| 20 |
+
pipe.load_lora(pipe.dit, "models/train/Qwen-Video-Edit_lora/epoch-4.safetensors")
|
| 21 |
+
|
| 22 |
+
video = pipe(edit_video=edit_video, prompts=prompts, height=640, width=384, num_frames=45, cfg_scale=4.0, num_inference_steps=40, seed=0)
|
| 23 |
+
save_video(video, "video_Qwen-Video-Edit-lora.mp4", fps=16)
|
examples/stable_diffusion/model_inference/stable-diffusion-v1-5.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
],
|
| 12 |
+
tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
|
| 13 |
+
)
|
| 14 |
+
|
| 15 |
+
image = pipe(
|
| 16 |
+
prompt="a photo of an astronaut riding a horse on mars, high quality, detailed",
|
| 17 |
+
negative_prompt="blurry, low quality, deformed",
|
| 18 |
+
cfg_scale=7.5,
|
| 19 |
+
height=512,
|
| 20 |
+
width=512,
|
| 21 |
+
seed=42,
|
| 22 |
+
rand_device="cuda",
|
| 23 |
+
num_inference_steps=50,
|
| 24 |
+
)
|
| 25 |
+
image.save("image.jpg")
|
examples/stable_diffusion/model_inference_low_vram/stable-diffusion-v1-5.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
|
| 4 |
+
|
| 5 |
+
vram_config = {
|
| 6 |
+
"offload_dtype": torch.float32,
|
| 7 |
+
"offload_device": "cpu",
|
| 8 |
+
"onload_dtype": torch.float32,
|
| 9 |
+
"onload_device": "cpu",
|
| 10 |
+
"preparing_dtype": torch.float32,
|
| 11 |
+
"preparing_device": "cuda",
|
| 12 |
+
"computation_dtype": torch.float32,
|
| 13 |
+
"computation_device": "cuda",
|
| 14 |
+
}
|
| 15 |
+
pipe = StableDiffusionPipeline.from_pretrained(
|
| 16 |
+
torch_dtype=torch.float32,
|
| 17 |
+
model_configs=[
|
| 18 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors", **vram_config),
|
| 19 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors", **vram_config),
|
| 20 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config),
|
| 21 |
+
],
|
| 22 |
+
tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
|
| 23 |
+
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
|
| 24 |
+
)
|
| 25 |
+
|
| 26 |
+
image = pipe(
|
| 27 |
+
prompt="a photo of an astronaut riding a horse on mars, high quality, detailed",
|
| 28 |
+
negative_prompt="blurry, low quality, deformed",
|
| 29 |
+
cfg_scale=7.5,
|
| 30 |
+
height=512,
|
| 31 |
+
width=512,
|
| 32 |
+
seed=42,
|
| 33 |
+
rand_device="cuda",
|
| 34 |
+
num_inference_steps=50,
|
| 35 |
+
)
|
| 36 |
+
image.save("image.jpg")
|
examples/stable_diffusion/model_training/full/stable-diffusion-v1-5.sh
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion/stable-diffusion-v1-5/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
accelerate launch examples/stable_diffusion/model_training/train.py \
|
| 4 |
+
--dataset_base_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5 \
|
| 5 |
+
--dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5/metadata.csv \
|
| 6 |
+
--height 512 \
|
| 7 |
+
--width 512 \
|
| 8 |
+
--dataset_repeat 50 \
|
| 9 |
+
--model_id_with_origin_paths "AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors" \
|
| 10 |
+
--learning_rate 1e-5 \
|
| 11 |
+
--num_epochs 2 \
|
| 12 |
+
--trainable_models "unet" \
|
| 13 |
+
--remove_prefix_in_ckpt "pipe.unet." \
|
| 14 |
+
--output_path "./models/train/stable-diffusion-v1-5_full" \
|
| 15 |
+
--use_gradient_checkpointing
|
examples/stable_diffusion/model_training/lora/stable-diffusion-v1-5.sh
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion/stable-diffusion-v1-5/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
accelerate launch examples/stable_diffusion/model_training/train.py \
|
| 4 |
+
--dataset_base_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5 \
|
| 5 |
+
--dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5/metadata.csv \
|
| 6 |
+
--height 512 \
|
| 7 |
+
--width 512 \
|
| 8 |
+
--dataset_repeat 50 \
|
| 9 |
+
--model_id_with_origin_paths "AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors" \
|
| 10 |
+
--learning_rate 1e-4 \
|
| 11 |
+
--num_epochs 5 \
|
| 12 |
+
--remove_prefix_in_ckpt "pipe.unet." \
|
| 13 |
+
--output_path "./models/train/stable-diffusion-v1-5_lora" \
|
| 14 |
+
--lora_base_model "unet" \
|
| 15 |
+
--lora_target_modules "" \
|
| 16 |
+
--lora_rank 32 \
|
| 17 |
+
--use_gradient_checkpointing
|
examples/stable_diffusion/model_training/special/split_training/stable-diffusion-v1-5.sh
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion/stable-diffusion-v1-5/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
# Stage 1: cache deterministic preprocessing outputs.
|
| 4 |
+
accelerate launch examples/stable_diffusion/model_training/train.py \
|
| 5 |
+
--dataset_base_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5 \
|
| 6 |
+
--dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion/stable-diffusion-v1-5/metadata.csv \
|
| 7 |
+
--height 512 \
|
| 8 |
+
--width 512 \
|
| 9 |
+
--dataset_repeat 1 \
|
| 10 |
+
--model_id_with_origin_paths AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors \
|
| 11 |
+
--learning_rate 1e-4 \
|
| 12 |
+
--num_epochs 5 \
|
| 13 |
+
--remove_prefix_in_ckpt pipe.unet. \
|
| 14 |
+
--output_path ./models/train/stable-diffusion-v1-5_split_cache \
|
| 15 |
+
--lora_base_model unet \
|
| 16 |
+
--lora_target_modules '' \
|
| 17 |
+
--lora_rank 32 \
|
| 18 |
+
--use_gradient_checkpointing \
|
| 19 |
+
--offload_models AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors \
|
| 20 |
+
--task sft:data_process
|
| 21 |
+
|
| 22 |
+
# Stage 2: train LoRA from the cached dataset.
|
| 23 |
+
accelerate launch examples/stable_diffusion/model_training/train.py \
|
| 24 |
+
--dataset_base_path ./models/train/stable-diffusion-v1-5_split_cache \
|
| 25 |
+
--height 512 \
|
| 26 |
+
--width 512 \
|
| 27 |
+
--dataset_repeat 50 \
|
| 28 |
+
--model_id_with_origin_paths AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:unet/diffusion_pytorch_model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors \
|
| 29 |
+
--learning_rate 1e-4 \
|
| 30 |
+
--num_epochs 5 \
|
| 31 |
+
--remove_prefix_in_ckpt pipe.unet. \
|
| 32 |
+
--output_path ./models/train/stable-diffusion-v1-5_split \
|
| 33 |
+
--lora_base_model unet \
|
| 34 |
+
--lora_target_modules '' \
|
| 35 |
+
--lora_rank 32 \
|
| 36 |
+
--use_gradient_checkpointing \
|
| 37 |
+
--offload_models AI-ModelScope/stable-diffusion-v1-5:text_encoder/model.safetensors,AI-ModelScope/stable-diffusion-v1-5:vae/diffusion_pytorch_model.safetensors \
|
| 38 |
+
--task sft:train
|
examples/stable_diffusion/model_training/special/split_training/validate.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
],
|
| 12 |
+
tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
|
| 13 |
+
)
|
| 14 |
+
pipe.load_lora(pipe.unet, './models/train/stable-diffusion-v1-5_split/epoch-4.safetensors')
|
| 15 |
+
|
| 16 |
+
image = pipe(
|
| 17 |
+
prompt="a dog",
|
| 18 |
+
negative_prompt="blurry, low quality, deformed",
|
| 19 |
+
cfg_scale=7.5,
|
| 20 |
+
height=512,
|
| 21 |
+
width=512,
|
| 22 |
+
seed=42,
|
| 23 |
+
rand_device="cuda",
|
| 24 |
+
num_inference_steps=50,
|
| 25 |
+
)
|
| 26 |
+
image.save('split_training_stable-diffusion-v1-5.jpg')
|
examples/stable_diffusion/model_training/train.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch, os, argparse, accelerate
|
| 2 |
+
from diffsynth.core import UnifiedDataset
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline, ModelConfig
|
| 4 |
+
from diffsynth.diffusion import *
|
| 5 |
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
class StableDiffusionTrainingModule(DiffusionTrainingModule):
|
| 9 |
+
def __init__(
|
| 10 |
+
self,
|
| 11 |
+
model_paths=None, model_id_with_origin_paths=None,
|
| 12 |
+
tokenizer_path=None,
|
| 13 |
+
trainable_models=None,
|
| 14 |
+
lora_base_model=None, lora_target_modules="", lora_rank=32, lora_checkpoint=None,
|
| 15 |
+
preset_lora_path=None, preset_lora_model=None,
|
| 16 |
+
use_gradient_checkpointing=True,
|
| 17 |
+
use_gradient_checkpointing_offload=False,
|
| 18 |
+
extra_inputs=None,
|
| 19 |
+
fp8_models=None,
|
| 20 |
+
offload_models=None,
|
| 21 |
+
quant_options=None,
|
| 22 |
+
resume_from_checkpoint=None, remove_prefix_in_ckpt=None,
|
| 23 |
+
device="cpu",
|
| 24 |
+
task="sft",
|
| 25 |
+
):
|
| 26 |
+
super().__init__()
|
| 27 |
+
# Load models
|
| 28 |
+
model_configs = self.parse_model_configs(model_paths, model_id_with_origin_paths, fp8_models=fp8_models, offload_models=offload_models, quant_options=quant_options, device=device)
|
| 29 |
+
tokenizer_config = self.parse_path_or_model_id(tokenizer_path, ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"))
|
| 30 |
+
self.pipe = StableDiffusionPipeline.from_pretrained(torch_dtype=torch.float32, device=device, model_configs=model_configs, tokenizer_config=tokenizer_config)
|
| 31 |
+
self.pipe = self.split_pipeline_units(task, self.pipe, trainable_models, lora_base_model)
|
| 32 |
+
self.resume_from_checkpoint(resume_from_checkpoint, remove_prefix_in_ckpt)
|
| 33 |
+
|
| 34 |
+
# Training mode
|
| 35 |
+
self.switch_pipe_to_training_mode(
|
| 36 |
+
self.pipe, trainable_models,
|
| 37 |
+
lora_base_model, lora_target_modules, lora_rank, lora_checkpoint,
|
| 38 |
+
preset_lora_path, preset_lora_model,
|
| 39 |
+
task=task,
|
| 40 |
+
)
|
| 41 |
+
|
| 42 |
+
# Other configs
|
| 43 |
+
self.use_gradient_checkpointing = use_gradient_checkpointing
|
| 44 |
+
self.use_gradient_checkpointing_offload = use_gradient_checkpointing_offload
|
| 45 |
+
self.extra_inputs = extra_inputs.split(",") if extra_inputs is not None else []
|
| 46 |
+
self.fp8_models = fp8_models
|
| 47 |
+
self.task = task
|
| 48 |
+
self.task_to_loss = {
|
| 49 |
+
"sft:data_process": lambda pipe, *args: args,
|
| 50 |
+
"direct_distill:data_process": lambda pipe, *args: args,
|
| 51 |
+
"sft": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
|
| 52 |
+
"sft:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
|
| 53 |
+
"direct_distill": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
|
| 54 |
+
"direct_distill:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
|
| 55 |
+
}
|
| 56 |
+
|
| 57 |
+
def get_pipeline_inputs(self, data):
|
| 58 |
+
inputs_posi = {"prompt": data["prompt"]}
|
| 59 |
+
inputs_nega = {"negative_prompt": ""}
|
| 60 |
+
inputs_shared = {
|
| 61 |
+
# Assume you are using this pipeline for inference,
|
| 62 |
+
# please fill in the input parameters.
|
| 63 |
+
"input_image": data["image"],
|
| 64 |
+
"height": data["image"].size[1],
|
| 65 |
+
"width": data["image"].size[0],
|
| 66 |
+
# Please do not modify the following parameters
|
| 67 |
+
# unless you clearly know what this will cause.
|
| 68 |
+
"cfg_scale": 1,
|
| 69 |
+
"rand_device": self.pipe.device,
|
| 70 |
+
"use_gradient_checkpointing": self.use_gradient_checkpointing,
|
| 71 |
+
"use_gradient_checkpointing_offload": self.use_gradient_checkpointing_offload,
|
| 72 |
+
}
|
| 73 |
+
inputs_shared = self.parse_extra_inputs(data, self.extra_inputs, inputs_shared)
|
| 74 |
+
return inputs_shared, inputs_posi, inputs_nega
|
| 75 |
+
|
| 76 |
+
def forward(self, data, inputs=None):
|
| 77 |
+
if inputs is None: inputs = self.get_pipeline_inputs(data)
|
| 78 |
+
inputs = self.transfer_data_to_device(inputs, self.pipe.device, self.pipe.torch_dtype)
|
| 79 |
+
for unit in self.pipe.units:
|
| 80 |
+
inputs = self.pipe.unit_runner(unit, self.pipe, *inputs)
|
| 81 |
+
loss = self.task_to_loss[self.task](self.pipe, *inputs)
|
| 82 |
+
return loss
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def parser():
|
| 86 |
+
parser = argparse.ArgumentParser(description="Simple example of a training script.")
|
| 87 |
+
parser = add_general_config(parser)
|
| 88 |
+
parser = add_image_size_config(parser)
|
| 89 |
+
parser.add_argument("--tokenizer_path", type=str, default=None, help="Path to tokenizer.")
|
| 90 |
+
return parser
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
if __name__ == "__main__":
|
| 94 |
+
parser = parser()
|
| 95 |
+
args = parser.parse_args()
|
| 96 |
+
accelerator = accelerate.Accelerator(
|
| 97 |
+
gradient_accumulation_steps=args.gradient_accumulation_steps,
|
| 98 |
+
kwargs_handlers=[accelerate.DistributedDataParallelKwargs(find_unused_parameters=args.find_unused_parameters)],
|
| 99 |
+
)
|
| 100 |
+
dataset = UnifiedDataset(
|
| 101 |
+
base_path=args.dataset_base_path,
|
| 102 |
+
metadata_path=args.dataset_metadata_path,
|
| 103 |
+
repeat=args.dataset_repeat,
|
| 104 |
+
data_file_keys=args.data_file_keys.split(","),
|
| 105 |
+
main_data_operator=UnifiedDataset.default_image_operator(
|
| 106 |
+
base_path=args.dataset_base_path,
|
| 107 |
+
max_pixels=args.max_pixels,
|
| 108 |
+
height=args.height,
|
| 109 |
+
width=args.width,
|
| 110 |
+
height_division_factor=32,
|
| 111 |
+
width_division_factor=32,
|
| 112 |
+
)
|
| 113 |
+
)
|
| 114 |
+
model = StableDiffusionTrainingModule(
|
| 115 |
+
model_paths=args.model_paths,
|
| 116 |
+
model_id_with_origin_paths=args.model_id_with_origin_paths,
|
| 117 |
+
tokenizer_path=args.tokenizer_path,
|
| 118 |
+
trainable_models=args.trainable_models,
|
| 119 |
+
lora_base_model=args.lora_base_model,
|
| 120 |
+
lora_target_modules=args.lora_target_modules,
|
| 121 |
+
lora_rank=args.lora_rank,
|
| 122 |
+
lora_checkpoint=args.lora_checkpoint,
|
| 123 |
+
preset_lora_path=args.preset_lora_path,
|
| 124 |
+
preset_lora_model=args.preset_lora_model,
|
| 125 |
+
use_gradient_checkpointing=args.use_gradient_checkpointing,
|
| 126 |
+
use_gradient_checkpointing_offload=args.use_gradient_checkpointing_offload,
|
| 127 |
+
extra_inputs=args.extra_inputs,
|
| 128 |
+
fp8_models=args.fp8_models,
|
| 129 |
+
offload_models=args.offload_models,
|
| 130 |
+
quant_options=args.quant_options,
|
| 131 |
+
resume_from_checkpoint=args.resume_from_checkpoint,
|
| 132 |
+
remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
|
| 133 |
+
task=args.task,
|
| 134 |
+
device="cpu" if args.enable_model_cpu_offload else accelerator.device,
|
| 135 |
+
)
|
| 136 |
+
model_logger = ModelLogger(
|
| 137 |
+
args.output_path,
|
| 138 |
+
remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
|
| 139 |
+
enable_tensorboard_log=args.enable_tensorboard_log,
|
| 140 |
+
enable_swanlab_log=args.enable_swanlab_log,
|
| 141 |
+
swanlab_project=args.swanlab_project,
|
| 142 |
+
enable_wandb_log=args.enable_wandb_log,
|
| 143 |
+
wandb_project=args.wandb_project,
|
| 144 |
+
enable_csv_log=args.enable_csv_log,
|
| 145 |
+
)
|
| 146 |
+
launcher_map = {
|
| 147 |
+
"sft:data_process": launch_data_process_task,
|
| 148 |
+
"direct_distill:data_process": launch_data_process_task,
|
| 149 |
+
"sft": launch_training_task,
|
| 150 |
+
"sft:train": launch_training_task,
|
| 151 |
+
"direct_distill": launch_training_task,
|
| 152 |
+
"direct_distill:train": launch_training_task,
|
| 153 |
+
}
|
| 154 |
+
launcher_map[args.task](accelerator, dataset, model, model_logger, args=args)
|
examples/stable_diffusion/model_training/validate_full/stable-diffusion-v1-5.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline, ModelConfig
|
| 2 |
+
from diffsynth.core import load_state_dict
|
| 3 |
+
import torch
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
],
|
| 12 |
+
tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
|
| 13 |
+
)
|
| 14 |
+
state_dict = load_state_dict("./models/train/stable-diffusion-v1-5_full/epoch-1.safetensors", torch_dtype=torch.float32)
|
| 15 |
+
pipe.unet.load_state_dict(state_dict)
|
| 16 |
+
|
| 17 |
+
image = pipe(
|
| 18 |
+
prompt="a dog",
|
| 19 |
+
negative_prompt="blurry, low quality, deformed",
|
| 20 |
+
cfg_scale=7.5,
|
| 21 |
+
height=512,
|
| 22 |
+
width=512,
|
| 23 |
+
seed=42,
|
| 24 |
+
rand_device="cuda",
|
| 25 |
+
num_inference_steps=50,
|
| 26 |
+
)
|
| 27 |
+
image.save("image_stable-diffusion-v1-5_full.jpg")
|
examples/stable_diffusion/model_training/validate_lora/stable-diffusion-v1-5.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion import StableDiffusionPipeline
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
],
|
| 12 |
+
tokenizer_config=ModelConfig(model_id="AI-ModelScope/stable-diffusion-v1-5", origin_file_pattern="tokenizer/"),
|
| 13 |
+
)
|
| 14 |
+
pipe.load_lora(pipe.unet, "models/train/stable-diffusion-v1-5_lora/epoch-4.safetensors")
|
| 15 |
+
|
| 16 |
+
image = pipe(
|
| 17 |
+
prompt="a dog",
|
| 18 |
+
negative_prompt="blurry, low quality, deformed",
|
| 19 |
+
cfg_scale=7.5,
|
| 20 |
+
height=512,
|
| 21 |
+
width=512,
|
| 22 |
+
seed=42,
|
| 23 |
+
rand_device="cuda",
|
| 24 |
+
num_inference_steps=50,
|
| 25 |
+
)
|
| 26 |
+
image.save("image_stable-diffusion-v1-5.jpg")
|
examples/stable_diffusion_xl/model_inference/stable-diffusion-xl-base-1.0.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionXLPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
|
| 14 |
+
tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
|
| 15 |
+
)
|
| 16 |
+
|
| 17 |
+
image = pipe(
|
| 18 |
+
prompt="a photo of an astronaut riding a horse on mars",
|
| 19 |
+
negative_prompt="",
|
| 20 |
+
cfg_scale=5.0,
|
| 21 |
+
height=1024,
|
| 22 |
+
width=1024,
|
| 23 |
+
seed=42,
|
| 24 |
+
num_inference_steps=50,
|
| 25 |
+
)
|
| 26 |
+
image.save("image.jpg")
|
examples/stable_diffusion_xl/model_inference_low_vram/stable-diffusion-xl-base-1.0.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
|
| 4 |
+
|
| 5 |
+
vram_config = {
|
| 6 |
+
"offload_dtype": torch.float32,
|
| 7 |
+
"offload_device": "cpu",
|
| 8 |
+
"onload_dtype": torch.float32,
|
| 9 |
+
"onload_device": "cpu",
|
| 10 |
+
"preparing_dtype": torch.float32,
|
| 11 |
+
"preparing_device": "cuda",
|
| 12 |
+
"computation_dtype": torch.float32,
|
| 13 |
+
"computation_device": "cuda",
|
| 14 |
+
}
|
| 15 |
+
pipe = StableDiffusionXLPipeline.from_pretrained(
|
| 16 |
+
torch_dtype=torch.float32,
|
| 17 |
+
model_configs=[
|
| 18 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors", **vram_config),
|
| 19 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors", **vram_config),
|
| 20 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors", **vram_config),
|
| 21 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors", **vram_config),
|
| 22 |
+
],
|
| 23 |
+
tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
|
| 24 |
+
tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
|
| 25 |
+
vram_limit=torch.cuda.mem_get_info("cuda")[1] / (1024 ** 3) - 0.5,
|
| 26 |
+
)
|
| 27 |
+
|
| 28 |
+
image = pipe(
|
| 29 |
+
prompt="a photo of an astronaut riding a horse on mars",
|
| 30 |
+
negative_prompt="",
|
| 31 |
+
cfg_scale=5.0,
|
| 32 |
+
height=1024,
|
| 33 |
+
width=1024,
|
| 34 |
+
seed=42,
|
| 35 |
+
num_inference_steps=50,
|
| 36 |
+
)
|
| 37 |
+
image.save("image.jpg")
|
examples/stable_diffusion_xl/model_training/full/stable-diffusion-xl-base-1.0.sh
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion_xl/stable-diffusion-xl-base-1.0/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
accelerate launch examples/stable_diffusion_xl/model_training/train.py \
|
| 4 |
+
--dataset_base_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0 \
|
| 5 |
+
--dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0/metadata.csv \
|
| 6 |
+
--height 1024 \
|
| 7 |
+
--width 1024 \
|
| 8 |
+
--dataset_repeat 10 \
|
| 9 |
+
--model_id_with_origin_paths "stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors" \
|
| 10 |
+
--learning_rate 1e-5 \
|
| 11 |
+
--num_epochs 2 \
|
| 12 |
+
--trainable_models "unet" \
|
| 13 |
+
--remove_prefix_in_ckpt "pipe.unet." \
|
| 14 |
+
--output_path "./models/train/stable-diffusion-xl-base-1.0_full" \
|
| 15 |
+
--use_gradient_checkpointing
|
examples/stable_diffusion_xl/model_training/lora/stable-diffusion-xl-base-1.0.sh
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion_xl/stable-diffusion-xl-base-1.0/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
accelerate launch examples/stable_diffusion_xl/model_training/train.py \
|
| 4 |
+
--dataset_base_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0 \
|
| 5 |
+
--dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0/metadata.csv \
|
| 6 |
+
--height 1024 \
|
| 7 |
+
--width 1024 \
|
| 8 |
+
--dataset_repeat 10 \
|
| 9 |
+
--model_id_with_origin_paths "stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors" \
|
| 10 |
+
--learning_rate 1e-4 \
|
| 11 |
+
--num_epochs 5 \
|
| 12 |
+
--remove_prefix_in_ckpt "pipe.unet." \
|
| 13 |
+
--output_path "./models/train/stable-diffusion-xl-base-1.0_lora" \
|
| 14 |
+
--lora_base_model "unet" \
|
| 15 |
+
--lora_target_modules "mid_block.attentions.0.proj_in,mid_block.attentions.0.proj_out,down_blocks.1.attentions.0.proj_in,down_blocks.1.attentions.0.proj_out,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.1.attentions.1.proj_in,down_blocks.1.attentions.1.proj_out,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.proj_in,down_blocks.2.attentions.0.proj_out,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.2,down_blocks.2.attentions.1.proj_in,down_blocks.2.attentions.1.proj_out,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.2,mid_block.attentions.0.transformer_blocks.0.attn1.to_k,mid_block.attentions.0.transformer_blocks.0.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn1.to_q,mid_block.attentions.0.transformer_blocks.0.attn1.to_v,mid_block.attentions.0.transformer_blocks.0.attn2.to_k,mid_block.attentions.0.transformer_blocks.0.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn2.to_q,mid_block.attentions.0.transformer_blocks.0.attn2.to_v,mid_block.attentions.0.transformer_blocks.0.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.0.ff.net.2,mid_block.attentions.0.transformer_blocks.1.attn1.to_k,mid_block.attentions.0.transformer_blocks.1.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn1.to_q,mid_block.attentions.0.transformer_blocks.1.attn1.to_v,mid_block.attentions.0.transformer_blocks.1.attn2.to_k,mid_block.attentions.0.transformer_blocks.1.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn2.to_q,mid_block.attentions.0.transformer_blocks.1.attn2.to_v,mid_block.attentions.0.transformer_blocks.1.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.1.ff.net.2,mid_block.attentions.0.transformer_blocks.2.attn1.to_k,mid_block.attentions.0.transformer_blocks.2.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn1.to_q,mid_block.attentions.0.transformer_blocks.2.attn1.to_v,mid_block.attentions.0.transformer_blocks.2.attn2.to_k,mid_block.attentions.0.transformer_blocks.2.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn2.to_q,mid_block.attentions.0.transformer_blocks.2.attn2.to_v,mid_block.attentions.0.transformer_blocks.2.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.2.ff.net.2,mid_block.attentions.0.transformer_blocks.3.attn1.to_k,mid_block.attentions.0.transformer_blocks.3.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn1.to_q,mid_block.attentions.0.transformer_blocks.3.attn1.to_v,mid_block.attentions.0.transformer_blocks.3.attn2.to_k,mid_block.attentions.0.transformer_blocks.3.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn2.to_q,mid_block.attentions.0.transformer_blocks.3.attn2.to_v,mid_block.attentions.0.transformer_blocks.3.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.3.ff.net.2,mid_block.attentions.0.transformer_blocks.4.attn1.to_k,mid_block.attentions.0.transformer_blocks.4.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn1.to_q,mid_block.attentions.0.transformer_blocks.4.attn1.to_v,mid_block.attentions.0.transformer_blocks.4.attn2.to_k,mid_block.attentions.0.transformer_blocks.4.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn2.to_q,mid_block.attentions.0.transformer_blocks.4.attn2.to_v,mid_block.attentions.0.transformer_blocks.4.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.4.ff.net.2,mid_block.attentions.0.transformer_blocks.5.attn1.to_k,mid_block.attentions.0.transformer_blocks.5.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn1.to_q,mid_block.attentions.0.transformer_blocks.5.attn1.to_v,mid_block.attentions.0.transformer_blocks.5.attn2.to_k,mid_block.attentions.0.transformer_blocks.5.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn2.to_q,mid_block.attentions.0.transformer_blocks.5.attn2.to_v,mid_block.attentions.0.transformer_blocks.5.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.5.ff.net.2,mid_block.attentions.0.transformer_blocks.6.attn1.to_k,mid_block.attentions.0.transformer_blocks.6.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn1.to_q,mid_block.attentions.0.transformer_blocks.6.attn1.to_v,mid_block.attentions.0.transformer_blocks.6.attn2.to_k,mid_block.attentions.0.transformer_blocks.6.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn2.to_q,mid_block.attentions.0.transformer_blocks.6.attn2.to_v,mid_block.attentions.0.transformer_blocks.6.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.6.ff.net.2,mid_block.attentions.0.transformer_blocks.7.attn1.to_k,mid_block.attentions.0.transformer_blocks.7.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn1.to_q,mid_block.attentions.0.transformer_blocks.7.attn1.to_v,mid_block.attentions.0.transformer_blocks.7.attn2.to_k,mid_block.attentions.0.transformer_blocks.7.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn2.to_q,mid_block.attentions.0.transformer_blocks.7.attn2.to_v,mid_block.attentions.0.transformer_blocks.7.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.7.ff.net.2,mid_block.attentions.0.transformer_blocks.8.attn1.to_k,mid_block.attentions.0.transformer_blocks.8.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn1.to_q,mid_block.attentions.0.transformer_blocks.8.attn1.to_v,mid_block.attentions.0.transformer_blocks.8.attn2.to_k,mid_block.attentions.0.transformer_blocks.8.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn2.to_q,mid_block.attentions.0.transformer_blocks.8.attn2.to_v,mid_block.attentions.0.transformer_blocks.8.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.8.ff.net.2,mid_block.attentions.0.transformer_blocks.9.attn1.to_k,mid_block.attentions.0.transformer_blocks.9.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn1.to_q,mid_block.attentions.0.transformer_blocks.9.attn1.to_v,mid_block.attentions.0.transformer_blocks.9.attn2.to_k,mid_block.attentions.0.transformer_blocks.9.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn2.to_q,mid_block.attentions.0.transformer_blocks.9.attn2.to_v,mid_block.attentions.0.transformer_blocks.9.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.0.proj_in,up_blocks.0.attentions.0.proj_out,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.1.proj_in,up_blocks.0.attentions.1.proj_out,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.2.proj_in,up_blocks.0.attentions.2.proj_out,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.2,up_blocks.1.attentions.0.proj_in,up_blocks.1.attentions.0.proj_out,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.1.proj_in,up_blocks.1.attentions.1.proj_out,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.2.proj_in,up_blocks.1.attentions.2.proj_out,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.2" \
|
| 16 |
+
--lora_rank 32 \
|
| 17 |
+
--use_gradient_checkpointing \
|
| 18 |
+
--align_to_opensource_format
|
examples/stable_diffusion_xl/model_training/special/split_training/stable-diffusion-xl-base-1.0.sh
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
modelscope download --dataset DiffSynth-Studio/diffsynth_example_dataset --include "stable_diffusion_xl/stable-diffusion-xl-base-1.0/*" --local_dir ./data/diffsynth_example_dataset
|
| 2 |
+
|
| 3 |
+
# Stage 1: cache deterministic preprocessing outputs.
|
| 4 |
+
accelerate launch examples/stable_diffusion_xl/model_training/train.py \
|
| 5 |
+
--dataset_base_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0 \
|
| 6 |
+
--dataset_metadata_path data/diffsynth_example_dataset/stable_diffusion_xl/stable-diffusion-xl-base-1.0/metadata.csv \
|
| 7 |
+
--height 1024 \
|
| 8 |
+
--width 1024 \
|
| 9 |
+
--dataset_repeat 1 \
|
| 10 |
+
--model_id_with_origin_paths stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors \
|
| 11 |
+
--learning_rate 1e-4 \
|
| 12 |
+
--num_epochs 5 \
|
| 13 |
+
--remove_prefix_in_ckpt pipe.unet. \
|
| 14 |
+
--output_path ./models/train/stable-diffusion-xl-base-1.0_split_cache \
|
| 15 |
+
--lora_base_model unet \
|
| 16 |
+
--lora_target_modules mid_block.attentions.0.proj_in,mid_block.attentions.0.proj_out,down_blocks.1.attentions.0.proj_in,down_blocks.1.attentions.0.proj_out,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.1.attentions.1.proj_in,down_blocks.1.attentions.1.proj_out,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.proj_in,down_blocks.2.attentions.0.proj_out,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.2,down_blocks.2.attentions.1.proj_in,down_blocks.2.attentions.1.proj_out,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.2,mid_block.attentions.0.transformer_blocks.0.attn1.to_k,mid_block.attentions.0.transformer_blocks.0.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn1.to_q,mid_block.attentions.0.transformer_blocks.0.attn1.to_v,mid_block.attentions.0.transformer_blocks.0.attn2.to_k,mid_block.attentions.0.transformer_blocks.0.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn2.to_q,mid_block.attentions.0.transformer_blocks.0.attn2.to_v,mid_block.attentions.0.transformer_blocks.0.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.0.ff.net.2,mid_block.attentions.0.transformer_blocks.1.attn1.to_k,mid_block.attentions.0.transformer_blocks.1.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn1.to_q,mid_block.attentions.0.transformer_blocks.1.attn1.to_v,mid_block.attentions.0.transformer_blocks.1.attn2.to_k,mid_block.attentions.0.transformer_blocks.1.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn2.to_q,mid_block.attentions.0.transformer_blocks.1.attn2.to_v,mid_block.attentions.0.transformer_blocks.1.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.1.ff.net.2,mid_block.attentions.0.transformer_blocks.2.attn1.to_k,mid_block.attentions.0.transformer_blocks.2.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn1.to_q,mid_block.attentions.0.transformer_blocks.2.attn1.to_v,mid_block.attentions.0.transformer_blocks.2.attn2.to_k,mid_block.attentions.0.transformer_blocks.2.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn2.to_q,mid_block.attentions.0.transformer_blocks.2.attn2.to_v,mid_block.attentions.0.transformer_blocks.2.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.2.ff.net.2,mid_block.attentions.0.transformer_blocks.3.attn1.to_k,mid_block.attentions.0.transformer_blocks.3.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn1.to_q,mid_block.attentions.0.transformer_blocks.3.attn1.to_v,mid_block.attentions.0.transformer_blocks.3.attn2.to_k,mid_block.attentions.0.transformer_blocks.3.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn2.to_q,mid_block.attentions.0.transformer_blocks.3.attn2.to_v,mid_block.attentions.0.transformer_blocks.3.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.3.ff.net.2,mid_block.attentions.0.transformer_blocks.4.attn1.to_k,mid_block.attentions.0.transformer_blocks.4.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn1.to_q,mid_block.attentions.0.transformer_blocks.4.attn1.to_v,mid_block.attentions.0.transformer_blocks.4.attn2.to_k,mid_block.attentions.0.transformer_blocks.4.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn2.to_q,mid_block.attentions.0.transformer_blocks.4.attn2.to_v,mid_block.attentions.0.transformer_blocks.4.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.4.ff.net.2,mid_block.attentions.0.transformer_blocks.5.attn1.to_k,mid_block.attentions.0.transformer_blocks.5.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn1.to_q,mid_block.attentions.0.transformer_blocks.5.attn1.to_v,mid_block.attentions.0.transformer_blocks.5.attn2.to_k,mid_block.attentions.0.transformer_blocks.5.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn2.to_q,mid_block.attentions.0.transformer_blocks.5.attn2.to_v,mid_block.attentions.0.transformer_blocks.5.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.5.ff.net.2,mid_block.attentions.0.transformer_blocks.6.attn1.to_k,mid_block.attentions.0.transformer_blocks.6.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn1.to_q,mid_block.attentions.0.transformer_blocks.6.attn1.to_v,mid_block.attentions.0.transformer_blocks.6.attn2.to_k,mid_block.attentions.0.transformer_blocks.6.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn2.to_q,mid_block.attentions.0.transformer_blocks.6.attn2.to_v,mid_block.attentions.0.transformer_blocks.6.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.6.ff.net.2,mid_block.attentions.0.transformer_blocks.7.attn1.to_k,mid_block.attentions.0.transformer_blocks.7.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn1.to_q,mid_block.attentions.0.transformer_blocks.7.attn1.to_v,mid_block.attentions.0.transformer_blocks.7.attn2.to_k,mid_block.attentions.0.transformer_blocks.7.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn2.to_q,mid_block.attentions.0.transformer_blocks.7.attn2.to_v,mid_block.attentions.0.transformer_blocks.7.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.7.ff.net.2,mid_block.attentions.0.transformer_blocks.8.attn1.to_k,mid_block.attentions.0.transformer_blocks.8.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn1.to_q,mid_block.attentions.0.transformer_blocks.8.attn1.to_v,mid_block.attentions.0.transformer_blocks.8.attn2.to_k,mid_block.attentions.0.transformer_blocks.8.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn2.to_q,mid_block.attentions.0.transformer_blocks.8.attn2.to_v,mid_block.attentions.0.transformer_blocks.8.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.8.ff.net.2,mid_block.attentions.0.transformer_blocks.9.attn1.to_k,mid_block.attentions.0.transformer_blocks.9.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn1.to_q,mid_block.attentions.0.transformer_blocks.9.attn1.to_v,mid_block.attentions.0.transformer_blocks.9.attn2.to_k,mid_block.attentions.0.transformer_blocks.9.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn2.to_q,mid_block.attentions.0.transformer_blocks.9.attn2.to_v,mid_block.attentions.0.transformer_blocks.9.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.0.proj_in,up_blocks.0.attentions.0.proj_out,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.1.proj_in,up_blocks.0.attentions.1.proj_out,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.2.proj_in,up_blocks.0.attentions.2.proj_out,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.2,up_blocks.1.attentions.0.proj_in,up_blocks.1.attentions.0.proj_out,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.1.proj_in,up_blocks.1.attentions.1.proj_out,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.2.proj_in,up_blocks.1.attentions.2.proj_out,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.2 \
|
| 17 |
+
--lora_rank 32 \
|
| 18 |
+
--use_gradient_checkpointing \
|
| 19 |
+
--align_to_opensource_format \
|
| 20 |
+
--offload_models stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors \
|
| 21 |
+
--task sft:data_process
|
| 22 |
+
|
| 23 |
+
# Stage 2: train LoRA from the cached dataset.
|
| 24 |
+
accelerate launch examples/stable_diffusion_xl/model_training/train.py \
|
| 25 |
+
--dataset_base_path ./models/train/stable-diffusion-xl-base-1.0_split_cache \
|
| 26 |
+
--height 1024 \
|
| 27 |
+
--width 1024 \
|
| 28 |
+
--dataset_repeat 10 \
|
| 29 |
+
--model_id_with_origin_paths stabilityai/stable-diffusion-xl-base-1.0:text_encoder/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:text_encoder_2/model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:unet/diffusion_pytorch_model.safetensors,stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors \
|
| 30 |
+
--learning_rate 1e-4 \
|
| 31 |
+
--num_epochs 5 \
|
| 32 |
+
--remove_prefix_in_ckpt pipe.unet. \
|
| 33 |
+
--output_path ./models/train/stable-diffusion-xl-base-1.0_split \
|
| 34 |
+
--lora_base_model unet \
|
| 35 |
+
--lora_target_modules mid_block.attentions.0.proj_in,mid_block.attentions.0.proj_out,down_blocks.1.attentions.0.proj_in,down_blocks.1.attentions.0.proj_out,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.1.attentions.1.proj_in,down_blocks.1.attentions.1.proj_out,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.proj_in,down_blocks.2.attentions.0.proj_out,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.0.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.0.transformer_blocks.9.ff.net.2,down_blocks.2.attentions.1.proj_in,down_blocks.2.attentions.1.proj_out,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.0.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.0.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.1.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.1.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.2.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.2.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.3.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.3.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.4.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.4.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.5.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.5.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.6.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.6.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.7.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.7.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.8.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.8.ff.net.2,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn1.to_v,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_k,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_out.0,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_q,down_blocks.2.attentions.1.transformer_blocks.9.attn2.to_v,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.0.proj,down_blocks.2.attentions.1.transformer_blocks.9.ff.net.2,mid_block.attentions.0.transformer_blocks.0.attn1.to_k,mid_block.attentions.0.transformer_blocks.0.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn1.to_q,mid_block.attentions.0.transformer_blocks.0.attn1.to_v,mid_block.attentions.0.transformer_blocks.0.attn2.to_k,mid_block.attentions.0.transformer_blocks.0.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.0.attn2.to_q,mid_block.attentions.0.transformer_blocks.0.attn2.to_v,mid_block.attentions.0.transformer_blocks.0.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.0.ff.net.2,mid_block.attentions.0.transformer_blocks.1.attn1.to_k,mid_block.attentions.0.transformer_blocks.1.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn1.to_q,mid_block.attentions.0.transformer_blocks.1.attn1.to_v,mid_block.attentions.0.transformer_blocks.1.attn2.to_k,mid_block.attentions.0.transformer_blocks.1.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.1.attn2.to_q,mid_block.attentions.0.transformer_blocks.1.attn2.to_v,mid_block.attentions.0.transformer_blocks.1.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.1.ff.net.2,mid_block.attentions.0.transformer_blocks.2.attn1.to_k,mid_block.attentions.0.transformer_blocks.2.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn1.to_q,mid_block.attentions.0.transformer_blocks.2.attn1.to_v,mid_block.attentions.0.transformer_blocks.2.attn2.to_k,mid_block.attentions.0.transformer_blocks.2.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.2.attn2.to_q,mid_block.attentions.0.transformer_blocks.2.attn2.to_v,mid_block.attentions.0.transformer_blocks.2.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.2.ff.net.2,mid_block.attentions.0.transformer_blocks.3.attn1.to_k,mid_block.attentions.0.transformer_blocks.3.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn1.to_q,mid_block.attentions.0.transformer_blocks.3.attn1.to_v,mid_block.attentions.0.transformer_blocks.3.attn2.to_k,mid_block.attentions.0.transformer_blocks.3.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.3.attn2.to_q,mid_block.attentions.0.transformer_blocks.3.attn2.to_v,mid_block.attentions.0.transformer_blocks.3.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.3.ff.net.2,mid_block.attentions.0.transformer_blocks.4.attn1.to_k,mid_block.attentions.0.transformer_blocks.4.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn1.to_q,mid_block.attentions.0.transformer_blocks.4.attn1.to_v,mid_block.attentions.0.transformer_blocks.4.attn2.to_k,mid_block.attentions.0.transformer_blocks.4.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.4.attn2.to_q,mid_block.attentions.0.transformer_blocks.4.attn2.to_v,mid_block.attentions.0.transformer_blocks.4.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.4.ff.net.2,mid_block.attentions.0.transformer_blocks.5.attn1.to_k,mid_block.attentions.0.transformer_blocks.5.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn1.to_q,mid_block.attentions.0.transformer_blocks.5.attn1.to_v,mid_block.attentions.0.transformer_blocks.5.attn2.to_k,mid_block.attentions.0.transformer_blocks.5.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.5.attn2.to_q,mid_block.attentions.0.transformer_blocks.5.attn2.to_v,mid_block.attentions.0.transformer_blocks.5.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.5.ff.net.2,mid_block.attentions.0.transformer_blocks.6.attn1.to_k,mid_block.attentions.0.transformer_blocks.6.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn1.to_q,mid_block.attentions.0.transformer_blocks.6.attn1.to_v,mid_block.attentions.0.transformer_blocks.6.attn2.to_k,mid_block.attentions.0.transformer_blocks.6.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.6.attn2.to_q,mid_block.attentions.0.transformer_blocks.6.attn2.to_v,mid_block.attentions.0.transformer_blocks.6.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.6.ff.net.2,mid_block.attentions.0.transformer_blocks.7.attn1.to_k,mid_block.attentions.0.transformer_blocks.7.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn1.to_q,mid_block.attentions.0.transformer_blocks.7.attn1.to_v,mid_block.attentions.0.transformer_blocks.7.attn2.to_k,mid_block.attentions.0.transformer_blocks.7.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.7.attn2.to_q,mid_block.attentions.0.transformer_blocks.7.attn2.to_v,mid_block.attentions.0.transformer_blocks.7.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.7.ff.net.2,mid_block.attentions.0.transformer_blocks.8.attn1.to_k,mid_block.attentions.0.transformer_blocks.8.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn1.to_q,mid_block.attentions.0.transformer_blocks.8.attn1.to_v,mid_block.attentions.0.transformer_blocks.8.attn2.to_k,mid_block.attentions.0.transformer_blocks.8.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.8.attn2.to_q,mid_block.attentions.0.transformer_blocks.8.attn2.to_v,mid_block.attentions.0.transformer_blocks.8.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.8.ff.net.2,mid_block.attentions.0.transformer_blocks.9.attn1.to_k,mid_block.attentions.0.transformer_blocks.9.attn1.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn1.to_q,mid_block.attentions.0.transformer_blocks.9.attn1.to_v,mid_block.attentions.0.transformer_blocks.9.attn2.to_k,mid_block.attentions.0.transformer_blocks.9.attn2.to_out.0,mid_block.attentions.0.transformer_blocks.9.attn2.to_q,mid_block.attentions.0.transformer_blocks.9.attn2.to_v,mid_block.attentions.0.transformer_blocks.9.ff.net.0.proj,mid_block.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.0.proj_in,up_blocks.0.attentions.0.proj_out,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.0.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.0.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.1.proj_in,up_blocks.0.attentions.1.proj_out,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.1.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.1.transformer_blocks.9.ff.net.2,up_blocks.0.attentions.2.proj_in,up_blocks.0.attentions.2.proj_out,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.1.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.2.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.2.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.3.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.3.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.4.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.4.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.5.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.5.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.6.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.6.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.7.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.7.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.8.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.8.ff.net.2,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn1.to_v,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_k,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_out.0,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_q,up_blocks.0.attentions.2.transformer_blocks.9.attn2.to_v,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.0.proj,up_blocks.0.attentions.2.transformer_blocks.9.ff.net.2,up_blocks.1.attentions.0.proj_in,up_blocks.1.attentions.0.proj_out,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.0.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.0.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.1.proj_in,up_blocks.1.attentions.1.proj_out,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.1.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.1.transformer_blocks.1.ff.net.2,up_blocks.1.attentions.2.proj_in,up_blocks.1.attentions.2.proj_out,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.0.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.0.ff.net.2,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn1.to_v,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_k,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_out.0,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_q,up_blocks.1.attentions.2.transformer_blocks.1.attn2.to_v,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.0.proj,up_blocks.1.attentions.2.transformer_blocks.1.ff.net.2 \
|
| 36 |
+
--lora_rank 32 \
|
| 37 |
+
--use_gradient_checkpointing \
|
| 38 |
+
--align_to_opensource_format \
|
| 39 |
+
--offload_models stabilityai/stable-diffusion-xl-base-1.0:vae/diffusion_pytorch_model.safetensors \
|
| 40 |
+
--task sft:train
|
examples/stable_diffusion_xl/model_training/special/split_training/validate.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionXLPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
|
| 14 |
+
tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.unet, './models/train/stable-diffusion-xl-base-1.0_split/epoch-4.safetensors')
|
| 17 |
+
|
| 18 |
+
image = pipe(
|
| 19 |
+
prompt="a dog",
|
| 20 |
+
negative_prompt="",
|
| 21 |
+
cfg_scale=7.0,
|
| 22 |
+
height=1024,
|
| 23 |
+
width=1024,
|
| 24 |
+
seed=42,
|
| 25 |
+
num_inference_steps=50,
|
| 26 |
+
)
|
| 27 |
+
image.save('split_training_stable-diffusion-xl.jpg')
|
examples/stable_diffusion_xl/model_training/train.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch, os, argparse, accelerate
|
| 2 |
+
from diffsynth.core import UnifiedDataset
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline, ModelConfig
|
| 4 |
+
from diffsynth.diffusion import *
|
| 5 |
+
from diffsynth.utils.lora.sdxl import SdxlLoRAConverter
|
| 6 |
+
os.environ["TOKENIZERS_PARALLELISM"] = "false"
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
class StableDiffusionXLTrainingModule(DiffusionTrainingModule):
|
| 10 |
+
def __init__(
|
| 11 |
+
self,
|
| 12 |
+
model_paths=None, model_id_with_origin_paths=None,
|
| 13 |
+
tokenizer_path=None,
|
| 14 |
+
trainable_models=None,
|
| 15 |
+
lora_base_model=None, lora_target_modules="", lora_rank=32, lora_checkpoint=None,
|
| 16 |
+
preset_lora_path=None, preset_lora_model=None,
|
| 17 |
+
use_gradient_checkpointing=True,
|
| 18 |
+
use_gradient_checkpointing_offload=False,
|
| 19 |
+
extra_inputs=None,
|
| 20 |
+
fp8_models=None,
|
| 21 |
+
offload_models=None,
|
| 22 |
+
quant_options=None,
|
| 23 |
+
resume_from_checkpoint=None, remove_prefix_in_ckpt=None,
|
| 24 |
+
device="cpu",
|
| 25 |
+
task="sft",
|
| 26 |
+
):
|
| 27 |
+
super().__init__()
|
| 28 |
+
# Load models
|
| 29 |
+
model_configs = self.parse_model_configs(model_paths, model_id_with_origin_paths, fp8_models=fp8_models, offload_models=offload_models, quant_options=quant_options, device=device)
|
| 30 |
+
tokenizer_config = self.parse_path_or_model_id(tokenizer_path, ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"))
|
| 31 |
+
tokenizer_2_config = self.parse_path_or_model_id(tokenizer_path, ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"))
|
| 32 |
+
self.pipe = StableDiffusionXLPipeline.from_pretrained(torch_dtype=torch.float32, device=device, model_configs=model_configs, tokenizer_config=tokenizer_config, tokenizer_2_config=tokenizer_2_config)
|
| 33 |
+
self.pipe = self.split_pipeline_units(task, self.pipe, trainable_models, lora_base_model)
|
| 34 |
+
self.resume_from_checkpoint(resume_from_checkpoint, remove_prefix_in_ckpt)
|
| 35 |
+
|
| 36 |
+
# Training mode
|
| 37 |
+
self.switch_pipe_to_training_mode(
|
| 38 |
+
self.pipe, trainable_models,
|
| 39 |
+
lora_base_model, lora_target_modules, lora_rank, lora_checkpoint,
|
| 40 |
+
preset_lora_path, preset_lora_model,
|
| 41 |
+
task=task,
|
| 42 |
+
)
|
| 43 |
+
|
| 44 |
+
# Other configs
|
| 45 |
+
self.use_gradient_checkpointing = use_gradient_checkpointing
|
| 46 |
+
self.use_gradient_checkpointing_offload = use_gradient_checkpointing_offload
|
| 47 |
+
self.extra_inputs = extra_inputs.split(",") if extra_inputs is not None else []
|
| 48 |
+
self.fp8_models = fp8_models
|
| 49 |
+
self.task = task
|
| 50 |
+
self.task_to_loss = {
|
| 51 |
+
"sft:data_process": lambda pipe, *args: args,
|
| 52 |
+
"direct_distill:data_process": lambda pipe, *args: args,
|
| 53 |
+
"sft": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
|
| 54 |
+
"sft:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: FlowMatchSFTLoss(pipe, **inputs_shared, **inputs_posi),
|
| 55 |
+
"direct_distill": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
|
| 56 |
+
"direct_distill:train": lambda pipe, inputs_shared, inputs_posi, inputs_nega: DirectDistillLoss(pipe, **inputs_shared, **inputs_posi),
|
| 57 |
+
}
|
| 58 |
+
|
| 59 |
+
def get_pipeline_inputs(self, data):
|
| 60 |
+
inputs_posi = {"prompt": data["prompt"]}
|
| 61 |
+
inputs_nega = {"negative_prompt": ""}
|
| 62 |
+
inputs_shared = {
|
| 63 |
+
# Assume you are using this pipeline for inference,
|
| 64 |
+
# please fill in the input parameters.
|
| 65 |
+
"input_image": data["image"],
|
| 66 |
+
"height": data["image"].size[1],
|
| 67 |
+
"width": data["image"].size[0],
|
| 68 |
+
# Please do not modify the following parameters
|
| 69 |
+
# unless you clearly know what this will cause.
|
| 70 |
+
"cfg_scale": 1,
|
| 71 |
+
"rand_device": self.pipe.device,
|
| 72 |
+
"use_gradient_checkpointing": self.use_gradient_checkpointing,
|
| 73 |
+
"use_gradient_checkpointing_offload": self.use_gradient_checkpointing_offload,
|
| 74 |
+
}
|
| 75 |
+
inputs_shared = self.parse_extra_inputs(data, self.extra_inputs, inputs_shared)
|
| 76 |
+
return inputs_shared, inputs_posi, inputs_nega
|
| 77 |
+
|
| 78 |
+
def forward(self, data, inputs=None):
|
| 79 |
+
if inputs is None: inputs = self.get_pipeline_inputs(data)
|
| 80 |
+
inputs = self.transfer_data_to_device(inputs, self.pipe.device, self.pipe.torch_dtype)
|
| 81 |
+
for unit in self.pipe.units:
|
| 82 |
+
inputs = self.pipe.unit_runner(unit, self.pipe, *inputs)
|
| 83 |
+
loss = self.task_to_loss[self.task](self.pipe, *inputs)
|
| 84 |
+
return loss
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def parser():
|
| 88 |
+
parser = argparse.ArgumentParser(description="Simple example of a training script.")
|
| 89 |
+
parser = add_general_config(parser)
|
| 90 |
+
parser = add_image_size_config(parser)
|
| 91 |
+
parser.add_argument("--tokenizer_path", type=str, default=None, help="Path to tokenizer.")
|
| 92 |
+
parser.add_argument("--tokenizer_2_path", type=str, default=None, help="Path to tokenizer 2.")
|
| 93 |
+
parser.add_argument("--align_to_opensource_format", default=False, action="store_true", help="Whether to align the lora format to opensource format.")
|
| 94 |
+
return parser
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
if __name__ == "__main__":
|
| 98 |
+
parser = parser()
|
| 99 |
+
args = parser.parse_args()
|
| 100 |
+
accelerator = accelerate.Accelerator(
|
| 101 |
+
gradient_accumulation_steps=args.gradient_accumulation_steps,
|
| 102 |
+
kwargs_handlers=[accelerate.DistributedDataParallelKwargs(find_unused_parameters=args.find_unused_parameters)],
|
| 103 |
+
)
|
| 104 |
+
dataset = UnifiedDataset(
|
| 105 |
+
base_path=args.dataset_base_path,
|
| 106 |
+
metadata_path=args.dataset_metadata_path,
|
| 107 |
+
repeat=args.dataset_repeat,
|
| 108 |
+
data_file_keys=args.data_file_keys.split(","),
|
| 109 |
+
main_data_operator=UnifiedDataset.default_image_operator(
|
| 110 |
+
base_path=args.dataset_base_path,
|
| 111 |
+
max_pixels=args.max_pixels,
|
| 112 |
+
height=args.height,
|
| 113 |
+
width=args.width,
|
| 114 |
+
height_division_factor=32,
|
| 115 |
+
width_division_factor=32,
|
| 116 |
+
)
|
| 117 |
+
)
|
| 118 |
+
model = StableDiffusionXLTrainingModule(
|
| 119 |
+
model_paths=args.model_paths,
|
| 120 |
+
model_id_with_origin_paths=args.model_id_with_origin_paths,
|
| 121 |
+
tokenizer_path=args.tokenizer_path,
|
| 122 |
+
trainable_models=args.trainable_models,
|
| 123 |
+
lora_base_model=args.lora_base_model,
|
| 124 |
+
lora_target_modules=args.lora_target_modules,
|
| 125 |
+
lora_rank=args.lora_rank,
|
| 126 |
+
lora_checkpoint=args.lora_checkpoint,
|
| 127 |
+
preset_lora_path=args.preset_lora_path,
|
| 128 |
+
preset_lora_model=args.preset_lora_model,
|
| 129 |
+
use_gradient_checkpointing=args.use_gradient_checkpointing,
|
| 130 |
+
use_gradient_checkpointing_offload=args.use_gradient_checkpointing_offload,
|
| 131 |
+
extra_inputs=args.extra_inputs,
|
| 132 |
+
fp8_models=args.fp8_models,
|
| 133 |
+
offload_models=args.offload_models,
|
| 134 |
+
quant_options=args.quant_options,
|
| 135 |
+
resume_from_checkpoint=args.resume_from_checkpoint,
|
| 136 |
+
remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
|
| 137 |
+
task=args.task,
|
| 138 |
+
device="cpu" if args.enable_model_cpu_offload else accelerator.device,
|
| 139 |
+
)
|
| 140 |
+
model_logger = ModelLogger(
|
| 141 |
+
args.output_path,
|
| 142 |
+
remove_prefix_in_ckpt=args.remove_prefix_in_ckpt,
|
| 143 |
+
state_dict_converter=SdxlLoRAConverter.align_to_opensource_format if args.align_to_opensource_format else lambda x:x,
|
| 144 |
+
enable_tensorboard_log=args.enable_tensorboard_log,
|
| 145 |
+
enable_swanlab_log=args.enable_swanlab_log,
|
| 146 |
+
swanlab_project=args.swanlab_project,
|
| 147 |
+
enable_wandb_log=args.enable_wandb_log,
|
| 148 |
+
wandb_project=args.wandb_project,
|
| 149 |
+
enable_csv_log=args.enable_csv_log,
|
| 150 |
+
)
|
| 151 |
+
launcher_map = {
|
| 152 |
+
"sft:data_process": launch_data_process_task,
|
| 153 |
+
"direct_distill:data_process": launch_data_process_task,
|
| 154 |
+
"sft": launch_training_task,
|
| 155 |
+
"sft:train": launch_training_task,
|
| 156 |
+
"direct_distill": launch_training_task,
|
| 157 |
+
"direct_distill:train": launch_training_task,
|
| 158 |
+
}
|
| 159 |
+
launcher_map[args.task](accelerator, dataset, model, model_logger, args=args)
|
examples/stable_diffusion_xl/model_training/validate_full/stable-diffusion-xl-base-1.0.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline, ModelConfig
|
| 2 |
+
from diffsynth.core import load_state_dict
|
| 3 |
+
import torch
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionXLPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
|
| 14 |
+
tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
|
| 15 |
+
)
|
| 16 |
+
state_dict = load_state_dict("./models/train/stable-diffusion-xl-base-1.0_full/epoch-1.safetensors", torch_dtype=torch.float32)
|
| 17 |
+
pipe.unet.load_state_dict(state_dict)
|
| 18 |
+
|
| 19 |
+
image = pipe(
|
| 20 |
+
prompt="a dog",
|
| 21 |
+
negative_prompt="",
|
| 22 |
+
cfg_scale=7.0,
|
| 23 |
+
height=1024,
|
| 24 |
+
width=1024,
|
| 25 |
+
seed=42,
|
| 26 |
+
num_inference_steps=50,
|
| 27 |
+
)
|
| 28 |
+
image.save("image_stable-diffusion-xl-base-1.0_full.jpg")
|
examples/stable_diffusion_xl/model_training/validate_lora/stable-diffusion-xl-base-1.0.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.core import ModelConfig
|
| 3 |
+
from diffsynth.pipelines.stable_diffusion_xl import StableDiffusionXLPipeline
|
| 4 |
+
|
| 5 |
+
pipe = StableDiffusionXLPipeline.from_pretrained(
|
| 6 |
+
torch_dtype=torch.float32,
|
| 7 |
+
model_configs=[
|
| 8 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder/model.safetensors"),
|
| 9 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="text_encoder_2/model.safetensors"),
|
| 10 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="unet/diffusion_pytorch_model.safetensors"),
|
| 11 |
+
ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
|
| 12 |
+
],
|
| 13 |
+
tokenizer_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer/"),
|
| 14 |
+
tokenizer_2_config=ModelConfig(model_id="stabilityai/stable-diffusion-xl-base-1.0", origin_file_pattern="tokenizer_2/"),
|
| 15 |
+
)
|
| 16 |
+
pipe.load_lora(pipe.unet, "models/train/stable-diffusion-xl-base-1.0_lora/epoch-4.safetensors")
|
| 17 |
+
|
| 18 |
+
image = pipe(
|
| 19 |
+
prompt="a dog",
|
| 20 |
+
negative_prompt="",
|
| 21 |
+
cfg_scale=7.0,
|
| 22 |
+
height=1024,
|
| 23 |
+
width=1024,
|
| 24 |
+
seed=42,
|
| 25 |
+
num_inference_steps=50,
|
| 26 |
+
)
|
| 27 |
+
image.save("image_stable-diffusion-xl-base-1.0.jpg")
|
examples/wanvideo/README.md
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
English Document: https://diffsynth-studio-doc.readthedocs.io/en/latest/Model_Details/Wan.html
|
| 2 |
+
|
| 3 |
+
中文文档:https://diffsynth-studio-doc.readthedocs.io/zh-cn/latest/Model_Details/Wan.html
|
examples/wanvideo/acceleration/Wan2.2-Animate-2-14B-usp.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
import torch.distributed as dist
|
| 3 |
+
from PIL import Image
|
| 4 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 5 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 6 |
+
from modelscope import dataset_snapshot_download
|
| 7 |
+
|
| 8 |
+
vram_config = {
|
| 9 |
+
"offload_dtype": torch.bfloat16,
|
| 10 |
+
"offload_device": "cpu",
|
| 11 |
+
"onload_dtype": torch.bfloat16,
|
| 12 |
+
"onload_device": "cuda",
|
| 13 |
+
"preparing_dtype": torch.bfloat16,
|
| 14 |
+
"preparing_device": "cuda",
|
| 15 |
+
"computation_dtype": torch.bfloat16,
|
| 16 |
+
"computation_device": "cuda",
|
| 17 |
+
}
|
| 18 |
+
|
| 19 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 20 |
+
torch_dtype=torch.bfloat16,
|
| 21 |
+
device="cuda",
|
| 22 |
+
use_usp=True,
|
| 23 |
+
model_configs=[
|
| 24 |
+
ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="wan_animate_2/wan_animate_2_bf16.safetensors", **vram_config),
|
| 25 |
+
ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="videomodel/Wan-AI/models_t5_umt5-xxl-enc-bf16.pth", **vram_config),
|
| 26 |
+
ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="videomodel/Wan-AI/models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth", **vram_config),
|
| 27 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="Wan2.1_VAE.pth", **vram_config),
|
| 28 |
+
],
|
| 29 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.2-Animate-2-14B", origin_file_pattern="videomodel/Wan-AI/umt5-xxl/"),
|
| 30 |
+
)
|
| 31 |
+
|
| 32 |
+
# Character animation: reference image (identity) + reference video (motion) -> animated video.
|
| 33 |
+
dataset_snapshot_download(
|
| 34 |
+
"DiffSynth-Studio/diffsynth_example_dataset",
|
| 35 |
+
local_dir="data/diffsynth_example_dataset",
|
| 36 |
+
allow_file_pattern="wanvideo/Wan2.2-Animate-2-14B/*"
|
| 37 |
+
)
|
| 38 |
+
reference_image = Image.open("data/diffsynth_example_dataset/wanvideo/Wan2.2-Animate-2-14B/refimage.jpg").convert("RGB")
|
| 39 |
+
reference_video = VideoData("data/diffsynth_example_dataset/wanvideo/Wan2.2-Animate-2-14B/refvideo.mp4").raw_data()
|
| 40 |
+
# Example 1: single-clip generation
|
| 41 |
+
num_frames = 81
|
| 42 |
+
video = pipe(
|
| 43 |
+
prompt="人物外观描述:一名长黑发女性,穿着白色半透明蕾丝长袖上衣,衣身带有花卉刺绣,下身搭配白色百褶短裙和黑色腰带,脚穿米白色厚底运动鞋。 背景描述:背景为现代室内空间,墙面和柜体以浅灰色为主,后方设有两扇深色落地窗或玻璃门,顶部安装长条形灯具,中央有一块浅色长方形台面。",
|
| 44 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 45 |
+
animate2_prompt_ref="视频中的人在做动作,背景静止",
|
| 46 |
+
animate2_reference_image=reference_image,
|
| 47 |
+
animate2_reference_video=reference_video[:num_frames],
|
| 48 |
+
animate2_offload_kv=True,
|
| 49 |
+
num_frames=num_frames, height=1280, width=720,
|
| 50 |
+
num_inference_steps=40, cfg_scale=3.0,
|
| 51 |
+
seed=0, tiled=True,
|
| 52 |
+
)
|
| 53 |
+
if dist.get_rank() == 0:
|
| 54 |
+
save_video(video, "video_Wan2.2-Animate-2-14B.mp4", fps=24, quality=5)
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
# Example 2: multi-clip long-video generation
|
| 58 |
+
def generate_long_video(pipe, reference_image, cond_images, clip_len, first_num=1, **kwargs):
|
| 59 |
+
assert clip_len > first_num, "clip_len must be greater than first_num"
|
| 60 |
+
|
| 61 |
+
def zigzag_padding(array, target_len):
|
| 62 |
+
if len(array) == 1:
|
| 63 |
+
return [array[0]] * target_len
|
| 64 |
+
idx, flip, out = 0, False, []
|
| 65 |
+
while len(out) < target_len:
|
| 66 |
+
out.append(array[idx])
|
| 67 |
+
idx += -1 if flip else 1
|
| 68 |
+
if idx == 0 or idx == len(array) - 1:
|
| 69 |
+
flip = not flip
|
| 70 |
+
return out[:target_len]
|
| 71 |
+
|
| 72 |
+
real_len = len(cond_images)
|
| 73 |
+
if real_len == 0:
|
| 74 |
+
return []
|
| 75 |
+
step = clip_len - first_num
|
| 76 |
+
# Precompute clip count so clips of `clip_len` stepping by `step` tile the (padded) driving video.
|
| 77 |
+
num_clips = 1 if real_len <= clip_len else (real_len - clip_len + step - 1) // step + 1
|
| 78 |
+
target_len = clip_len + (num_clips - 1) * step
|
| 79 |
+
if real_len < target_len:
|
| 80 |
+
cond_images = zigzag_padding(cond_images, target_len)
|
| 81 |
+
|
| 82 |
+
all_frames = []
|
| 83 |
+
prev_tail = None
|
| 84 |
+
for i in range(num_clips):
|
| 85 |
+
start = i * step
|
| 86 |
+
seg_driving = cond_images[start:start + clip_len]
|
| 87 |
+
seg_out = pipe(
|
| 88 |
+
animate2_reference_image=reference_image,
|
| 89 |
+
animate2_reference_video=seg_driving,
|
| 90 |
+
animate2_refert_images=None if i == 0 else prev_tail,
|
| 91 |
+
num_frames=clip_len,
|
| 92 |
+
**kwargs,
|
| 93 |
+
)
|
| 94 |
+
prev_tail = seg_out[-first_num:]
|
| 95 |
+
if i != 0:
|
| 96 |
+
seg_out = seg_out[first_num:]
|
| 97 |
+
all_frames.extend(seg_out)
|
| 98 |
+
return all_frames[:real_len]
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
clip_len = 81
|
| 102 |
+
long_video = generate_long_video(
|
| 103 |
+
pipe,
|
| 104 |
+
reference_image=reference_image,
|
| 105 |
+
cond_images=reference_video,
|
| 106 |
+
clip_len=clip_len,
|
| 107 |
+
first_num=1,
|
| 108 |
+
prompt="人物外观描述:一名长黑发女性,穿着白色半透明蕾丝长袖上衣,衣身带有花卉刺绣,下身搭配白色百褶短裙和黑色腰带,脚穿米白色厚底运动鞋。 背景描述:背景为现代室内空间,墙面和柜体以浅灰色为主,后方设有两扇深色落地窗或玻璃门,顶部安装长条形灯具,中央有一块浅色长方形台面。",
|
| 109 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 110 |
+
animate2_prompt_ref="视频中的人在做动作,背景静止",
|
| 111 |
+
animate2_offload_kv=True,
|
| 112 |
+
height=1280, width=720,
|
| 113 |
+
num_inference_steps=40, cfg_scale=3.0,
|
| 114 |
+
seed=0, tiled=True,
|
| 115 |
+
)
|
| 116 |
+
if dist.get_rank() == 0:
|
| 117 |
+
save_video(long_video, "video_Wan2.2-Animate-2-14B-long.mp4", fps=24, quality=5)
|
examples/wanvideo/acceleration/unified_sequence_parallel.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
import torch.distributed as dist
|
| 6 |
+
|
| 7 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 8 |
+
torch_dtype=torch.bfloat16,
|
| 9 |
+
device="cuda",
|
| 10 |
+
use_usp=True,
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
],
|
| 16 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 17 |
+
)
|
| 18 |
+
|
| 19 |
+
# Text-to-video
|
| 20 |
+
video = pipe(
|
| 21 |
+
prompt="一名宇航员身穿太空服,面朝镜头骑着一匹机械马在火星表面驰骋。红色的荒凉地表延伸至远方,点缀着巨大的陨石坑和奇特的岩石结构。机械马的步伐稳健,扬起微弱的尘埃,展现出未来科技与原始探索的完美结合。宇航员手持操控装置,目光坚定,仿佛正在开辟人类的新疆域。背景是深邃的宇宙和蔚蓝的地球,画面既科幻又充满希望,让人不禁畅想未来的星际生活。",
|
| 22 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 23 |
+
seed=0, tiled=True,
|
| 24 |
+
)
|
| 25 |
+
if dist.get_rank() == 0:
|
| 26 |
+
save_video(video, "video1.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/LongCat-Video.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 3 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 7 |
+
torch_dtype=torch.bfloat16,
|
| 8 |
+
device="cuda",
|
| 9 |
+
model_configs=[
|
| 10 |
+
ModelConfig(model_id="meituan-longcat/LongCat-Video", origin_file_pattern="dit/diffusion_pytorch_model*.safetensors"),
|
| 11 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 12 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-14B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 13 |
+
],
|
| 14 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 15 |
+
)
|
| 16 |
+
|
| 17 |
+
# Text-to-video
|
| 18 |
+
video = pipe(
|
| 19 |
+
prompt="In a realistic photography style, a white boy around seven or eight years old sits on a park bench, wearing a light blue T-shirt, denim shorts, and white sneakers. He holds an ice cream cone with vanilla and chocolate flavors, and beside him is a medium-sized golden Labrador. Smiling, the boy offers the ice cream to the dog, who eagerly licks it with its tongue. The sun is shining brightly, and the background features a green lawn and several tall trees, creating a warm and loving scene.",
|
| 20 |
+
negative_prompt="Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards",
|
| 21 |
+
seed=0, tiled=True, num_frames=93,
|
| 22 |
+
cfg_scale=2, sigma_shift=1,
|
| 23 |
+
)
|
| 24 |
+
save_video(video, "video_1_LongCat-Video.mp4", fps=15, quality=5)
|
| 25 |
+
|
| 26 |
+
# Video-continuation (The number of frames in `longcat_video` should be 4n+1.)
|
| 27 |
+
longcat_video = video[-17:]
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="In a realistic photography style, a white boy around seven or eight years old sits on a park bench, wearing a light blue T-shirt, denim shorts, and white sneakers. He holds an ice cream cone with vanilla and chocolate flavors, and beside him is a medium-sized golden Labrador. Smiling, the boy offers the ice cream to the dog, who eagerly licks it with its tongue. The sun is shining brightly, and the background features a green lawn and several tall trees, creating a warm and loving scene.",
|
| 30 |
+
negative_prompt="Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards",
|
| 31 |
+
seed=1, tiled=True, num_frames=93,
|
| 32 |
+
cfg_scale=2, sigma_shift=1,
|
| 33 |
+
longcat_video=longcat_video,
|
| 34 |
+
)
|
| 35 |
+
save_video(video, "video_2_LongCat-Video.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Video-As-Prompt-Wan2.1-14B.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
import PIL
|
| 3 |
+
from PIL import Image
|
| 4 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 5 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 6 |
+
from modelscope import dataset_snapshot_download
|
| 7 |
+
from typing import List
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 11 |
+
torch_dtype=torch.bfloat16,
|
| 12 |
+
device="cuda",
|
| 13 |
+
model_configs=[
|
| 14 |
+
ModelConfig(model_id="ByteDance/Video-As-Prompt-Wan2.1-14B", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
|
| 15 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-I2V-14B-720P", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 16 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-I2V-14B-720P", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 17 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-I2V-14B-720P", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 18 |
+
],
|
| 19 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 20 |
+
)
|
| 21 |
+
|
| 22 |
+
dataset_snapshot_download("DiffSynth-Studio/example_video_dataset", allow_file_pattern="wanvap/*", local_dir="data/example_video_dataset")
|
| 23 |
+
ref_video_path = 'data/example_video_dataset/wanvap/vap_ref.mp4'
|
| 24 |
+
target_image_path = 'data/example_video_dataset/wanvap/input_image.jpg'
|
| 25 |
+
|
| 26 |
+
def select_frames(video_frames, num):
|
| 27 |
+
idx = torch.linspace(0, len(video_frames) - 1, num).long().tolist()
|
| 28 |
+
return [video_frames[i] for i in idx]
|
| 29 |
+
|
| 30 |
+
image = Image.open(target_image_path).convert("RGB")
|
| 31 |
+
ref_video = VideoData(ref_video_path, height=480, width=832)
|
| 32 |
+
ref_frames = select_frames(ref_video, num=49)
|
| 33 |
+
|
| 34 |
+
vap_prompt = "A man stands with his back to the camera on a dirt path overlooking sun-drenched, rolling green tea plantations. He wears a blue and green plaid shirt, dark pants, and white shoes. As he turns to face the camera and spreads his arms, a brief, magical burst of sparkling golden light particles envelops him. Through this shimmer, he seamlessly transforms into a Labubu toy character. His head morphs into the iconic large, furry-eared head of the toy, featuring a wide grin with pointed teeth and red cheek markings. The character retains the man's original plaid shirt and clothing, which now fit its stylized, cartoonish body. The camera remains static throughout the transformation, positioned low among the tea bushes, maintaining a consistent view of the subject and the expansive scenery."
|
| 35 |
+
prompt = "A young woman with curly hair, wearing a green hijab and a floral dress, plays a violin in front of a vintage green car on a tree-lined street. She executes a swift counter-clockwise turn to face the camera. During the turn, a brilliant shower of golden, sparkling particles erupts and momentarily obscures her figure. As the particles fade, she is revealed to have seamlessly transformed into a Labubu toy character. This new figure, now with the toy's signature large ears, big eyes, and toothy grin, maintains the original pose and continues playing the violin. The character's clothing—the green hijab, floral dress, and black overcoat—remains identical to the woman's. Throughout this transition, the camera stays static, and the street-side environment remains completely consistent."
|
| 36 |
+
negative_prompt = "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards"
|
| 37 |
+
|
| 38 |
+
video = pipe(
|
| 39 |
+
prompt=prompt,
|
| 40 |
+
negative_prompt=negative_prompt,
|
| 41 |
+
input_image=image,
|
| 42 |
+
seed=42, tiled=True,
|
| 43 |
+
height=480, width=832,
|
| 44 |
+
num_frames=49,
|
| 45 |
+
vap_video=ref_frames,
|
| 46 |
+
vap_prompt=vap_prompt,
|
| 47 |
+
negative_vap_prompt=negative_prompt,
|
| 48 |
+
)
|
| 49 |
+
save_video(video, "video_Video-As-Prompt-Wan2.1-14B.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan-Dancer-14B-global.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="global_model.safetensors"),
|
| 13 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
dataset_snapshot_download(
|
| 20 |
+
"DiffSynth-Studio/diffsynth_example_dataset",
|
| 21 |
+
local_dir="data/diffsynth_example_dataset",
|
| 22 |
+
allow_file_pattern="wanvideo/Wan-Dancer-14B-global/*"
|
| 23 |
+
)
|
| 24 |
+
# This is a specialized model with the following constraints on its input parameters:
|
| 25 |
+
# * The model outputs a sequence of keyframes rather than a video; therefore, `framewise_decoding=True` must be set.
|
| 26 |
+
# * When the number of keyframes is $n$, `num_frames` = 4 * (n - 1) + 1.
|
| 27 |
+
# * Reducing `height`, `width`, `num_frames`, or `num_inference_steps` may lead to severe artifacts or generation failure.
|
| 28 |
+
# * The audio file specified by `wantodance_music_path` must match the video duration, calculated as (`num_frames` / 7.5) seconds.
|
| 29 |
+
# * The width and height of `wantodance_reference_image` must be multiples of 16.
|
| 30 |
+
# * `wantodance_fps` is configurable, but since the model appears to have been trained exclusively at 7.5 FPS, setting it to other values is not recommended.
|
| 31 |
+
# * The first frame of `wantodance_keyframes` is the `wantodance_reference_image`, while all subsequent frames are solid black.
|
| 32 |
+
# * `wantodance_keyframes_mask` indicates the positions of valid frames within `wantodance_keyframes`.
|
| 33 |
+
wantodance_keyframes = VideoData("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-global/keyframes.mp4")
|
| 34 |
+
wantodance_keyframes = [wantodance_keyframes[i] for i in range(149)]
|
| 35 |
+
video = pipe(
|
| 36 |
+
prompt="一个人正在跳舞,舞蹈种类是韩舞。帧率是7.5000",
|
| 37 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 38 |
+
seed=0, tiled=False,
|
| 39 |
+
height=1280, width=720, num_frames=149,
|
| 40 |
+
num_inference_steps=48,
|
| 41 |
+
wantodance_music_path="data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-global/music.WAV",
|
| 42 |
+
wantodance_reference_image=Image.open("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-global/refimage.jpg"),
|
| 43 |
+
wantodance_fps=7.5,
|
| 44 |
+
wantodance_keyframes=wantodance_keyframes,
|
| 45 |
+
wantodance_keyframes_mask=[1] + [0] * 148,
|
| 46 |
+
framewise_decoding=True,
|
| 47 |
+
)
|
| 48 |
+
save_video(video, "video_Wan-Dancer-14B-global.mp4", fps=7.5, quality=5)
|
examples/wanvideo/model_inference/Wan-Dancer-14B-local.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch, os
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="local_model.safetensors"),
|
| 13 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="Wan-AI/Wan-Dancer-14B", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
dataset_snapshot_download(
|
| 20 |
+
"DiffSynth-Studio/diffsynth_example_dataset",
|
| 21 |
+
local_dir="data/diffsynth_example_dataset",
|
| 22 |
+
allow_file_pattern="wanvideo/Wan-Dancer-14B-local/*"
|
| 23 |
+
)
|
| 24 |
+
# This is a specialized model with the following constraints on its input parameters:
|
| 25 |
+
# * The model renders and outputs video based on a sequence of keyframes; therefore, `wantodance_keyframes` must be provided correctly.
|
| 26 |
+
# * If you need to generate a long video, please generate it in segments, and ensure that `wantodance_music_path`, `wantodance_keyframes`, and `wantodance_keyframes_mask` are properly split accordingly.
|
| 27 |
+
# * The audio file specified by `wantodance_music_path` must match the video duration, calculated as (`num_frames` / 30) seconds.
|
| 28 |
+
# * The width and height of `wantodance_reference_image` must be multiples of 16.
|
| 29 |
+
# * `wantodance_fps` is configurable, but since the model appears to have been trained exclusively at 30 FPS, setting it to other values is not recommended.
|
| 30 |
+
# * In `wantodance_keyframes`, frames that are not keyframes should be solid black.
|
| 31 |
+
# * `wantodance_keyframes_mask` indicates the positions of valid frames within `wantodance_keyframes`.
|
| 32 |
+
wantodance_keyframes = VideoData("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-local/keyframes.mp4")
|
| 33 |
+
wantodance_keyframes = [wantodance_keyframes[i] for i in range(149)]
|
| 34 |
+
video = pipe(
|
| 35 |
+
prompt="一个人正在跳舞,舞蹈种类是古典舞,图像清晰程度高,人物动作平均幅度中等,人物动作最大幅度中等。, 帧率是30fps。",
|
| 36 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 37 |
+
seed=0, tiled=True,
|
| 38 |
+
height=1280, width=720, num_frames=149,
|
| 39 |
+
num_inference_steps=24,
|
| 40 |
+
wantodance_music_path="data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-local/music.wav",
|
| 41 |
+
wantodance_reference_image=Image.open("data/diffsynth_example_dataset/wanvideo/Wan-Dancer-14B-local/refimage.jpg"),
|
| 42 |
+
wantodance_fps=30,
|
| 43 |
+
wantodance_keyframes=wantodance_keyframes,
|
| 44 |
+
wantodance_keyframes_mask=[1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
| 45 |
+
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
| 46 |
+
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
| 47 |
+
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
| 48 |
+
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
| 49 |
+
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
| 50 |
+
1],
|
| 51 |
+
)
|
| 52 |
+
save_video(video, "video_Wan-Dancer-14B-local.mp4", fps=30, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-1.3b-speedcontrol-v1.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 8 |
+
torch_dtype=torch.bfloat16,
|
| 9 |
+
device="cuda",
|
| 10 |
+
model_configs=[
|
| 11 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 12 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 13 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 14 |
+
ModelConfig(model_id="DiffSynth-Studio/Wan2.1-1.3b-speedcontrol-v1", origin_file_pattern="model.safetensors"),
|
| 15 |
+
],
|
| 16 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 17 |
+
)
|
| 18 |
+
|
| 19 |
+
# Text-to-video
|
| 20 |
+
video = pipe(
|
| 21 |
+
prompt="纪实摄影风格画面,一只活泼的小狗在绿茵茵的草地上迅速奔跑。小狗毛色棕黄,两只耳朵立起,神情专注而欢快。阳光洒在它身上,使得毛发看上去格外柔软而闪亮。背景是一片开阔的草地,偶尔点缀着几朵野花,远处隐约可见蓝天和几片白云。透视感鲜明,捕捉小狗奔跑时的动感和四周草地的生机。中景侧面移动视角。",
|
| 22 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 23 |
+
seed=1, tiled=True,
|
| 24 |
+
motion_bucket_id=0
|
| 25 |
+
)
|
| 26 |
+
save_video(video, "video_slow_Wan2.1-1.3b-speedcontrol-v1.mp4", fps=15, quality=5)
|
| 27 |
+
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="纪实摄影风格画面,一只活泼的小狗在绿茵茵的草地上迅速奔跑。小狗毛色棕黄,两只耳朵立起,神情专注而欢快。阳光洒在它身上,使得毛发看上去格外柔软而闪亮。背景是一片开阔的草地,偶尔点缀着几朵野花,远处隐约可见蓝天和几片白云。透视感鲜明,捕捉小狗奔跑时的动感和四周草地的生机。中景侧面移动视角。",
|
| 30 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 31 |
+
seed=1, tiled=True,
|
| 32 |
+
motion_bucket_id=100
|
| 33 |
+
)
|
| 34 |
+
save_video(video, "video_fast_Wan2.1-1.3b-speedcontrol-v1.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-FLF2V-14B-720P.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="Wan-AI/Wan2.1-FLF2V-14B-720P", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
|
| 20 |
+
dataset_snapshot_download(
|
| 21 |
+
dataset_id="DiffSynth-Studio/examples_in_diffsynth",
|
| 22 |
+
local_dir="./",
|
| 23 |
+
allow_file_pattern=["data/examples/wan/first_frame.jpeg", "data/examples/wan/last_frame.jpeg"]
|
| 24 |
+
)
|
| 25 |
+
|
| 26 |
+
# First and last frame to video
|
| 27 |
+
video = pipe(
|
| 28 |
+
prompt="写实风格,一个女生手持枯萎的花站在花园中,镜头逐渐拉远,记录下花园的全貌。",
|
| 29 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 30 |
+
input_image=Image.open("data/examples/wan/first_frame.jpeg").resize((960, 960)),
|
| 31 |
+
end_image=Image.open("data/examples/wan/last_frame.jpeg").resize((960, 960)),
|
| 32 |
+
seed=0, tiled=True,
|
| 33 |
+
height=960, width=960, num_frames=33,
|
| 34 |
+
sigma_shift=16,
|
| 35 |
+
)
|
| 36 |
+
save_video(video, "video_Wan2.1-FLF2V-14B-720P.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-Control.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-Control", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
|
| 20 |
+
dataset_snapshot_download(
|
| 21 |
+
dataset_id="DiffSynth-Studio/examples_in_diffsynth",
|
| 22 |
+
local_dir="./",
|
| 23 |
+
allow_file_pattern=f"data/examples/wan/control_video.mp4"
|
| 24 |
+
)
|
| 25 |
+
|
| 26 |
+
# Control video
|
| 27 |
+
control_video = VideoData("data/examples/wan/control_video.mp4", height=832, width=576)
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="扁平风格动漫,一位长发少女优雅起舞。她五官精致,大眼睛明亮有神,黑色长发柔顺光泽。身穿淡蓝色T恤和深蓝色牛仔短裤。背景是粉色。",
|
| 30 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 31 |
+
control_video=control_video, height=832, width=576, num_frames=49,
|
| 32 |
+
seed=1, tiled=True
|
| 33 |
+
)
|
| 34 |
+
save_video(video, "video_Wan2.1-Fun-1.3B-Control.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-Fun-1.3B-InP.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-1.3B-InP", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
|
| 20 |
+
dataset_snapshot_download(
|
| 21 |
+
dataset_id="DiffSynth-Studio/examples_in_diffsynth",
|
| 22 |
+
local_dir="./",
|
| 23 |
+
allow_file_pattern=f"data/examples/wan/input_image.jpg"
|
| 24 |
+
)
|
| 25 |
+
image = Image.open("data/examples/wan/input_image.jpg")
|
| 26 |
+
|
| 27 |
+
# First and last frame to video
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
|
| 30 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 31 |
+
input_image=image,
|
| 32 |
+
seed=0, tiled=True
|
| 33 |
+
# You can input `end_image=xxx` to control the last frame of the video.
|
| 34 |
+
# The model will automatically generate the dynamic content between `input_image` and `end_image`.
|
| 35 |
+
)
|
| 36 |
+
save_video(video, "video_Wan2.1-Fun-1.3B-InP.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-Fun-14B-Control.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-Control", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
|
| 20 |
+
dataset_snapshot_download(
|
| 21 |
+
dataset_id="DiffSynth-Studio/examples_in_diffsynth",
|
| 22 |
+
local_dir="./",
|
| 23 |
+
allow_file_pattern=f"data/examples/wan/control_video.mp4"
|
| 24 |
+
)
|
| 25 |
+
|
| 26 |
+
# Control video
|
| 27 |
+
control_video = VideoData("data/examples/wan/control_video.mp4", height=832, width=576)
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="扁平风格动漫,一位长发少女优雅起舞。她五官精致,大眼睛明亮有神,黑色长发柔顺光泽。身穿淡蓝色T恤和深蓝色牛仔短裤。背景是粉色。",
|
| 30 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 31 |
+
control_video=control_video, height=832, width=576, num_frames=49,
|
| 32 |
+
seed=1, tiled=True
|
| 33 |
+
)
|
| 34 |
+
save_video(video, "video_Wan2.1-Fun-14B-Control.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-Fun-14B-InP.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-14B-InP", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
|
| 20 |
+
dataset_snapshot_download(
|
| 21 |
+
dataset_id="DiffSynth-Studio/examples_in_diffsynth",
|
| 22 |
+
local_dir="./",
|
| 23 |
+
allow_file_pattern=f"data/examples/wan/input_image.jpg"
|
| 24 |
+
)
|
| 25 |
+
image = Image.open("data/examples/wan/input_image.jpg")
|
| 26 |
+
|
| 27 |
+
# First and last frame to video
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
|
| 30 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 31 |
+
input_image=image,
|
| 32 |
+
seed=0, tiled=True
|
| 33 |
+
# You can input `end_image=xxx` to control the last frame of the video.
|
| 34 |
+
# The model will automatically generate the dynamic content between `input_image` and `end_image`.
|
| 35 |
+
)
|
| 36 |
+
save_video(video, "video_Wan2.1-Fun-14B-InP.mp4", fps=15, quality=5)
|
examples/wanvideo/model_inference/Wan2.1-Fun-V1.1-1.3B-Control-Camera.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import torch
|
| 2 |
+
from PIL import Image
|
| 3 |
+
from diffsynth.utils.data import save_video, VideoData
|
| 4 |
+
from diffsynth.pipelines.wan_video import WanVideoPipeline, ModelConfig
|
| 5 |
+
from modelscope import dataset_snapshot_download
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
pipe = WanVideoPipeline.from_pretrained(
|
| 9 |
+
torch_dtype=torch.bfloat16,
|
| 10 |
+
device="cuda",
|
| 11 |
+
model_configs=[
|
| 12 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="diffusion_pytorch_model*.safetensors"),
|
| 13 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="models_t5_umt5-xxl-enc-bf16.pth"),
|
| 14 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="Wan2.1_VAE.pth"),
|
| 15 |
+
ModelConfig(model_id="PAI/Wan2.1-Fun-V1.1-1.3B-Control-Camera", origin_file_pattern="models_clip_open-clip-xlm-roberta-large-vit-huge-14.pth"),
|
| 16 |
+
],
|
| 17 |
+
tokenizer_config=ModelConfig(model_id="Wan-AI/Wan2.1-T2V-1.3B", origin_file_pattern="google/umt5-xxl/"),
|
| 18 |
+
)
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
dataset_snapshot_download(
|
| 22 |
+
dataset_id="DiffSynth-Studio/examples_in_diffsynth",
|
| 23 |
+
local_dir="./",
|
| 24 |
+
allow_file_pattern=f"data/examples/wan/input_image.jpg"
|
| 25 |
+
)
|
| 26 |
+
input_image = Image.open("data/examples/wan/input_image.jpg")
|
| 27 |
+
|
| 28 |
+
video = pipe(
|
| 29 |
+
prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
|
| 30 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 31 |
+
seed=0, tiled=True,
|
| 32 |
+
input_image=input_image,
|
| 33 |
+
camera_control_direction="Left", camera_control_speed=0.01,
|
| 34 |
+
)
|
| 35 |
+
save_video(video, "video_left_Wan2.1-Fun-V1.1-1.3B-Control-Camera.mp4", fps=15, quality=5)
|
| 36 |
+
|
| 37 |
+
video = pipe(
|
| 38 |
+
prompt="一艘小船正勇敢地乘风破浪前行。蔚蓝的大海波涛汹涌,白色的浪花拍打着船身,但小船毫不畏惧,坚定地驶向远方。阳光洒在水面上,闪烁着金色的光芒,为这壮丽的场景增添了一抹温暖。镜头拉近,可以看到船上的旗帜迎风飘扬,象征着不屈的精神与冒险的勇气。这段画面充满力量,激励人心,展现了面对挑战时的无畏与执着。",
|
| 39 |
+
negative_prompt="色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
|
| 40 |
+
seed=0, tiled=True,
|
| 41 |
+
input_image=input_image,
|
| 42 |
+
camera_control_direction="Up", camera_control_speed=0.01,
|
| 43 |
+
)
|
| 44 |
+
save_video(video, "video_up_Wan2.1-Fun-V1.1-1.3B-Control-Camera.mp4", fps=15, quality=5)
|