support eligenv2 and context_control

2026-03-22 16:50:47 +00:00 · 2025-08-20 22:48:34 +08:00
parent 9ec0652339
commit 5e6f9f89f1
12 changed files with 371 additions and 2 deletions
--- a/examples/qwen_image/model_training/lora/Qwen-Image-Context-Control.sh
+++ b/examples/qwen_image/model_training/lora/Qwen-Image-Context-Control.sh
@@ -0,0 +1,20 @@
+accelerate launch examples/qwen_image/model_training/train.py \
+  --dataset_base_path "data/example_image_dataset" \
+  --dataset_metadata_path data/example_image_dataset/metadata_qwenimage_context.csv \
+  --data_file_keys "image,context_image" \
+  --max_pixels 1048576 \
+  --dataset_repeat 50 \
+  --model_id_with_origin_paths "Qwen/Qwen-Image:transformer/diffusion_pytorch_model*.safetensors,Qwen/Qwen-Image:text_encoder/model*.safetensors,Qwen/Qwen-Image:vae/diffusion_pytorch_model.safetensors" \
+  --learning_rate 1e-4 \
+  --num_epochs 5 \
+  --remove_prefix_in_ckpt "pipe.dit." \
+  --output_path "./models/train/Qwen-Image-Context-Control_lora" \
+  --lora_base_model "dit" \
+  --lora_target_modules "to_q,to_k,to_v,add_q_proj,add_k_proj,add_v_proj,to_out.0,to_add_out,img_mlp.net.2,img_mod.1,txt_mlp.net.2,txt_mod.1" \
+  --lora_rank 64 \
+  --lora_checkpoint "models/DiffSynth-Studio/Qwen-Image-Context-Control/model.safetensors" \
+  --extra_inputs "context_image" \
+  --use_gradient_checkpointing \
+  --find_unused_parameters
+
+# if you want to train from scratch, you can remove the --lora_checkpoint argument
--- a/examples/qwen_image/model_training/validate_lora/Qwen-Image-Context-Control.py
+++ b/examples/qwen_image/model_training/validate_lora/Qwen-Image-Context-Control.py
@@ -0,0 +1,19 @@
+from PIL import Image
+import torch
+from diffsynth.pipelines.qwen_image import QwenImagePipeline, ModelConfig
+
+pipe = QwenImagePipeline.from_pretrained(
+    torch_dtype=torch.bfloat16,
+    device="cuda",
+    model_configs=[
+        ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="transformer/diffusion_pytorch_model*.safetensors"),
+        ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="text_encoder/model*.safetensors"),
+        ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="vae/diffusion_pytorch_model.safetensors"),
+    ],
+    tokenizer_config=ModelConfig(model_id="Qwen/Qwen-Image", origin_file_pattern="tokenizer/"),
+)
+pipe.load_lora(pipe.dit, "models/train/Qwen-Image-Context-Control_lora/epoch-4.safetensors")
+image = Image.open("data/example_image_dataset/canny/image_1.jpg").resize((1024, 1024))
+prompt = "Context_Control. a dog"
+image = pipe(prompt=prompt, seed=0, context_image=image, height=1024, width=1024)
+image.save("image_context.jpg")