Upload setup_pod_vaelith.sh with huggingface_hub
Browse files- setup_pod_vaelith.sh +62 -0
setup_pod_vaelith.sh
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# Run this on the RunPod pod terminal (after you've created the pod yourself).
|
| 3 |
+
# Assumes an RTX 4090 (24GB) or similar, Ubuntu + CUDA image, network volume at /workspace.
|
| 4 |
+
# Network volume should be at least 100GB this time -- kyrael's pod hit disk-full twice at 50GB.
|
| 5 |
+
set -e
|
| 6 |
+
|
| 7 |
+
cd /workspace
|
| 8 |
+
|
| 9 |
+
echo "=== 1. Clone musubi-tuner ==="
|
| 10 |
+
git clone https://github.com/kohya-ss/musubi-tuner
|
| 11 |
+
cd musubi-tuner
|
| 12 |
+
pip install -e .
|
| 13 |
+
pip install transformers accelerate qwen-vl-utils "huggingface_hub[cli]"
|
| 14 |
+
|
| 15 |
+
echo "=== 2. Log into HuggingFace ==="
|
| 16 |
+
echo "Paste your token when prompted -- do NOT put it directly on the command line."
|
| 17 |
+
hf auth login
|
| 18 |
+
|
| 19 |
+
echo "=== 3. Download the RAW Krea2 model (~24.5GB, gated -- must have accepted access on huggingface.co/krea/Krea-2-Raw) ==="
|
| 20 |
+
mkdir -p /workspace/models
|
| 21 |
+
hf download krea/Krea-2-Raw raw.safetensors --local-dir /workspace/models/krea2_raw
|
| 22 |
+
|
| 23 |
+
echo "=== 4. Download VAE + text encoder ==="
|
| 24 |
+
hf download Comfy-Org/Qwen-Image_ComfyUI split_files/vae/qwen_image_vae.safetensors --local-dir /workspace/models/vae
|
| 25 |
+
hf download Comfy-Org/Qwen3-VL text_encoders/qwen3vl_4b_bf16.safetensors --local-dir /workspace/models/text_encoder
|
| 26 |
+
|
| 27 |
+
echo "=== 5. Download vaelith dataset ==="
|
| 28 |
+
mkdir -p /workspace/dataset
|
| 29 |
+
hf download JBARU/vaelith-dataset --repo-type dataset --local-dir /workspace/dataset/vaelith
|
| 30 |
+
|
| 31 |
+
echo "=== 6. Caption dataset (Qwen2.5-VL-7B) ==="
|
| 32 |
+
python /workspace/caption_dataset.py /workspace/dataset/vaelith vaelith
|
| 33 |
+
|
| 34 |
+
echo "=== 6b. Clean up captioning model cache (~16GB) -- this is what caused the disk-full crashes on kyrael's run ==="
|
| 35 |
+
rm -rf /workspace/.cache
|
| 36 |
+
df -h /workspace
|
| 37 |
+
|
| 38 |
+
VAE=/workspace/models/vae/split_files/vae/qwen_image_vae.safetensors
|
| 39 |
+
TE=/workspace/models/text_encoder/text_encoders/qwen3vl_4b_bf16.safetensors
|
| 40 |
+
DIT=/workspace/models/krea2_raw/raw.safetensors
|
| 41 |
+
|
| 42 |
+
echo "=== 7. Pre-cache latents + text encoder outputs (vaelith) ==="
|
| 43 |
+
python src/musubi_tuner/krea2_cache_latents.py --dataset_config /workspace/dataset_vaelith.toml --vae "$VAE"
|
| 44 |
+
python src/musubi_tuner/krea2_cache_text_encoder_outputs.py --dataset_config /workspace/dataset_vaelith.toml --text_encoder "$TE" --batch_size 1
|
| 45 |
+
|
| 46 |
+
echo "=== 8. Train vaelith LoRA ==="
|
| 47 |
+
echo "Using num_repeats=3 this time (kyrael used 10, which caused overtraining/rigidity)."
|
| 48 |
+
PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \
|
| 49 |
+
accelerate launch --num_cpu_threads_per_process 1 --mixed_precision bf16 \
|
| 50 |
+
src/musubi_tuner/krea2_train_network.py \
|
| 51 |
+
--dit "$DIT" --vae "$VAE" \
|
| 52 |
+
--dataset_config /workspace/dataset_vaelith.toml \
|
| 53 |
+
--sdpa --mixed_precision bf16 --fp8_base --fp8_scaled \
|
| 54 |
+
--timestep_sampling shift --weighting_scheme none --discrete_flow_shift 2.5 \
|
| 55 |
+
--optimizer_type adamw8bit --learning_rate 1e-4 --gradient_checkpointing \
|
| 56 |
+
--max_data_loader_n_workers 2 --persistent_data_loader_workers \
|
| 57 |
+
--network_module networks.lora_krea2 --network_dim 32 --network_alpha 16 \
|
| 58 |
+
--max_train_epochs 16 --save_every_n_epochs 2 --seed 42 \
|
| 59 |
+
--output_dir /workspace/output/vaelith --output_name vaelith_lora
|
| 60 |
+
|
| 61 |
+
echo "=== Done. LoRA is in /workspace/output/vaelith ==="
|
| 62 |
+
echo "Back it up immediately with: hf upload <your-username>/vaelith-lora /workspace/output/vaelith --repo-type model --private"
|