JBARU commited on
Commit
5f99e05
·
verified ·
1 Parent(s): 99d7da6

Upload setup_pod_vaelith.sh with huggingface_hub

Browse files
Files changed (1) hide show
  1. setup_pod_vaelith.sh +62 -0
setup_pod_vaelith.sh ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/bin/bash
2
+ # Run this on the RunPod pod terminal (after you've created the pod yourself).
3
+ # Assumes an RTX 4090 (24GB) or similar, Ubuntu + CUDA image, network volume at /workspace.
4
+ # Network volume should be at least 100GB this time -- kyrael's pod hit disk-full twice at 50GB.
5
+ set -e
6
+
7
+ cd /workspace
8
+
9
+ echo "=== 1. Clone musubi-tuner ==="
10
+ git clone https://github.com/kohya-ss/musubi-tuner
11
+ cd musubi-tuner
12
+ pip install -e .
13
+ pip install transformers accelerate qwen-vl-utils "huggingface_hub[cli]"
14
+
15
+ echo "=== 2. Log into HuggingFace ==="
16
+ echo "Paste your token when prompted -- do NOT put it directly on the command line."
17
+ hf auth login
18
+
19
+ echo "=== 3. Download the RAW Krea2 model (~24.5GB, gated -- must have accepted access on huggingface.co/krea/Krea-2-Raw) ==="
20
+ mkdir -p /workspace/models
21
+ hf download krea/Krea-2-Raw raw.safetensors --local-dir /workspace/models/krea2_raw
22
+
23
+ echo "=== 4. Download VAE + text encoder ==="
24
+ hf download Comfy-Org/Qwen-Image_ComfyUI split_files/vae/qwen_image_vae.safetensors --local-dir /workspace/models/vae
25
+ hf download Comfy-Org/Qwen3-VL text_encoders/qwen3vl_4b_bf16.safetensors --local-dir /workspace/models/text_encoder
26
+
27
+ echo "=== 5. Download vaelith dataset ==="
28
+ mkdir -p /workspace/dataset
29
+ hf download JBARU/vaelith-dataset --repo-type dataset --local-dir /workspace/dataset/vaelith
30
+
31
+ echo "=== 6. Caption dataset (Qwen2.5-VL-7B) ==="
32
+ python /workspace/caption_dataset.py /workspace/dataset/vaelith vaelith
33
+
34
+ echo "=== 6b. Clean up captioning model cache (~16GB) -- this is what caused the disk-full crashes on kyrael's run ==="
35
+ rm -rf /workspace/.cache
36
+ df -h /workspace
37
+
38
+ VAE=/workspace/models/vae/split_files/vae/qwen_image_vae.safetensors
39
+ TE=/workspace/models/text_encoder/text_encoders/qwen3vl_4b_bf16.safetensors
40
+ DIT=/workspace/models/krea2_raw/raw.safetensors
41
+
42
+ echo "=== 7. Pre-cache latents + text encoder outputs (vaelith) ==="
43
+ python src/musubi_tuner/krea2_cache_latents.py --dataset_config /workspace/dataset_vaelith.toml --vae "$VAE"
44
+ python src/musubi_tuner/krea2_cache_text_encoder_outputs.py --dataset_config /workspace/dataset_vaelith.toml --text_encoder "$TE" --batch_size 1
45
+
46
+ echo "=== 8. Train vaelith LoRA ==="
47
+ echo "Using num_repeats=3 this time (kyrael used 10, which caused overtraining/rigidity)."
48
+ PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \
49
+ accelerate launch --num_cpu_threads_per_process 1 --mixed_precision bf16 \
50
+ src/musubi_tuner/krea2_train_network.py \
51
+ --dit "$DIT" --vae "$VAE" \
52
+ --dataset_config /workspace/dataset_vaelith.toml \
53
+ --sdpa --mixed_precision bf16 --fp8_base --fp8_scaled \
54
+ --timestep_sampling shift --weighting_scheme none --discrete_flow_shift 2.5 \
55
+ --optimizer_type adamw8bit --learning_rate 1e-4 --gradient_checkpointing \
56
+ --max_data_loader_n_workers 2 --persistent_data_loader_workers \
57
+ --network_module networks.lora_krea2 --network_dim 32 --network_alpha 16 \
58
+ --max_train_epochs 16 --save_every_n_epochs 2 --seed 42 \
59
+ --output_dir /workspace/output/vaelith --output_name vaelith_lora
60
+
61
+ echo "=== Done. LoRA is in /workspace/output/vaelith ==="
62
+ echo "Back it up immediately with: hf upload <your-username>/vaelith-lora /workspace/output/vaelith --repo-type model --private"